Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8aa9038226 | ||
|
|
ba533dc8ae | ||
|
|
eb91a337cd | ||
|
|
f3ea6b301c | ||
|
|
51bf5326d9 | ||
|
|
8aae261018 | ||
|
|
8a2028bfa7 | ||
|
|
8421a34d8d | ||
|
|
c4cd460754 | ||
|
|
47cdf72404 | ||
|
|
f2d57420fb | ||
|
|
150a241a0f | ||
|
|
24f1f2c648 | ||
|
|
c2e1976659 | ||
|
|
eca6eb3f3b | ||
|
|
a80de88078 | ||
|
|
bb26edf351 | ||
|
|
a572eb2164 | ||
|
|
4ded8cff3d |
@@ -0,0 +1,62 @@
|
|||||||
|
name: CI
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches: [main, "feat/**", "fix/**"]
|
||||||
|
pull_request:
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
shell:
|
||||||
|
name: shell (shellcheck + syntax)
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: bash syntax check
|
||||||
|
run: |
|
||||||
|
fail=0
|
||||||
|
while IFS= read -r f; do
|
||||||
|
bash -n "$f" || { echo "::error file=$f::bash syntax error"; fail=1; }
|
||||||
|
done < <(find . -name '*.sh' -not -path './.git/*')
|
||||||
|
exit $fail
|
||||||
|
|
||||||
|
# Pinned to a release tag, not @master: a third-party action on a moving
|
||||||
|
# branch runs whatever that branch contains at the time CI fires.
|
||||||
|
- name: shellcheck
|
||||||
|
uses: ludeeus/action-shellcheck@2.0.0
|
||||||
|
env:
|
||||||
|
SHELLCHECK_OPTS: -S warning -e SC1091
|
||||||
|
|
||||||
|
python:
|
||||||
|
name: python ${{ matrix.python }}
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
strategy:
|
||||||
|
fail-fast: false
|
||||||
|
matrix:
|
||||||
|
# TrueNAS SCALE middleware runs 3.11+; keep the patch importable across
|
||||||
|
# the versions it may be injected into.
|
||||||
|
python: ["3.11", "3.12", "3.13"]
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- uses: actions/setup-python@v5
|
||||||
|
with:
|
||||||
|
python-version: ${{ matrix.python }}
|
||||||
|
|
||||||
|
- name: install dev deps
|
||||||
|
run: python -m pip install --upgrade pip pytest ruff
|
||||||
|
|
||||||
|
- name: ruff
|
||||||
|
run: ruff check patch tests tools
|
||||||
|
|
||||||
|
- name: pytest
|
||||||
|
run: pytest tests -v
|
||||||
|
|
||||||
|
- name: verify injected middleware blocks compile
|
||||||
|
# Belt-and-braces: the *_BLOCK strings are appended into live middlewared
|
||||||
|
# modules. A syntax error there would break the box at boot.
|
||||||
|
run: pytest tests/test_apply_blocks.py -v
|
||||||
@@ -0,0 +1,102 @@
|
|||||||
|
name: Release
|
||||||
|
|
||||||
|
# Push a tag, get a release. The body always comes from CHANGELOG.md, so there is
|
||||||
|
# no second place to write release notes and therefore no second place for them to
|
||||||
|
# go stale.
|
||||||
|
#
|
||||||
|
# git tag -a v0.4.0 -m "v0.4.0" && git push origin v0.4.0
|
||||||
|
#
|
||||||
|
# workflow_dispatch re-cuts (or updates) the release for a tag that already
|
||||||
|
# exists, since re-pushing an existing tag triggers nothing.
|
||||||
|
#
|
||||||
|
# It checks out the TAG, because the tagged code is what people install and it has
|
||||||
|
# to pass its own tests. That means it only works for tags that actually contain
|
||||||
|
# this tooling (>= v0.3.0). Tags older than that were backfilled by hand.
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
tags: ["v*"]
|
||||||
|
workflow_dispatch:
|
||||||
|
inputs:
|
||||||
|
tag:
|
||||||
|
description: "Existing tag to create a release for (e.g. v0.2.1)"
|
||||||
|
required: true
|
||||||
|
type: string
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: write
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
release:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- name: Resolve tag
|
||||||
|
id: tag
|
||||||
|
run: |
|
||||||
|
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||||
|
echo "tag=${{ inputs.tag }}" >> "$GITHUB_OUTPUT"
|
||||||
|
else
|
||||||
|
echo "tag=${GITHUB_REF#refs/tags/}" >> "$GITHUB_OUTPUT"
|
||||||
|
fi
|
||||||
|
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
with:
|
||||||
|
ref: ${{ steps.tag.outputs.tag }}
|
||||||
|
fetch-depth: 0
|
||||||
|
|
||||||
|
- uses: actions/setup-python@v5
|
||||||
|
with:
|
||||||
|
python-version: "3.13"
|
||||||
|
|
||||||
|
# Never publish a release for code that does not pass its own tests. A
|
||||||
|
# tagged commit is what people install; it has to be at least as good as
|
||||||
|
# main.
|
||||||
|
- name: install dev deps
|
||||||
|
run: python -m pip install --upgrade pip pytest ruff
|
||||||
|
|
||||||
|
- name: ruff
|
||||||
|
run: ruff check patch tests tools
|
||||||
|
|
||||||
|
- name: pytest
|
||||||
|
run: pytest tests -q
|
||||||
|
|
||||||
|
- name: shell syntax
|
||||||
|
run: |
|
||||||
|
fail=0
|
||||||
|
while IFS= read -r f; do
|
||||||
|
bash -n "$f" || { echo "::error file=$f::bash syntax error"; fail=1; }
|
||||||
|
done < <(find . -name '*.sh' -not -path './.git/*')
|
||||||
|
exit $fail
|
||||||
|
|
||||||
|
# Catches the failure mode this repo actually had: VERSION= drifted to
|
||||||
|
# three different values across the scripts, and nothing noticed.
|
||||||
|
- name: version matches tag and CHANGELOG has a section
|
||||||
|
run: python3 tools/release_notes.py check "${{ steps.tag.outputs.tag }}"
|
||||||
|
|
||||||
|
- name: extract release notes from CHANGELOG
|
||||||
|
run: |
|
||||||
|
python3 tools/release_notes.py notes "${{ steps.tag.outputs.tag }}" > /tmp/notes.md
|
||||||
|
echo "--- release body ---"
|
||||||
|
cat /tmp/notes.md
|
||||||
|
|
||||||
|
- name: create or update the release
|
||||||
|
env:
|
||||||
|
GH_TOKEN: ${{ github.token }}
|
||||||
|
TAG: ${{ steps.tag.outputs.tag }}
|
||||||
|
run: |
|
||||||
|
# Pre-1.0 and any -rc/-beta suffix ship as prereleases, not "Latest".
|
||||||
|
prerelease=""
|
||||||
|
case "$TAG" in
|
||||||
|
*-rc*|*-beta*|*-alpha*) prerelease="--prerelease" ;;
|
||||||
|
esac
|
||||||
|
|
||||||
|
if gh release view "$TAG" >/dev/null 2>&1; then
|
||||||
|
echo "Release $TAG exists — updating notes."
|
||||||
|
gh release edit "$TAG" --notes-file /tmp/notes.md
|
||||||
|
else
|
||||||
|
# shellcheck disable=SC2086
|
||||||
|
gh release create "$TAG" \
|
||||||
|
--title "$TAG" \
|
||||||
|
--notes-file /tmp/notes.md \
|
||||||
|
$prerelease
|
||||||
|
fi
|
||||||
@@ -4,3 +4,12 @@
|
|||||||
/apply.log.2
|
/apply.log.2
|
||||||
/hook_status.json
|
/hook_status.json
|
||||||
/disabled
|
/disabled
|
||||||
|
/nested_snapshots_enabled
|
||||||
|
|
||||||
|
# Python
|
||||||
|
__pycache__/
|
||||||
|
*.py[cod]
|
||||||
|
.pytest_cache/
|
||||||
|
.ruff_cache/
|
||||||
|
.venv/
|
||||||
|
venv/
|
||||||
|
|||||||
+284
@@ -1,5 +1,289 @@
|
|||||||
# Changelog
|
# Changelog
|
||||||
|
|
||||||
|
## v0.3.2 — 2026-07-13
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **`install.sh --disable-nested-snapshots` did not actually disable anything
|
||||||
|
until the next reboot.** `apply.sh` only ever *added* patches — there was no
|
||||||
|
revert path. Disabling removed the opt-in marker and then merely *skipped*
|
||||||
|
re-applying, but the overlay persists for the whole boot, so the previously
|
||||||
|
patched `plugins/cloud/{snapshot,crud}.py`, `plugins/cloud_backup/sync.py` and
|
||||||
|
`_truecloud_nested.py` were all still sitting there — and middlewared
|
||||||
|
re-imported them on the restart `install.sh` performs.
|
||||||
|
|
||||||
|
It printed *"DISABLED (stock guard restored)"* while the feature kept running.
|
||||||
|
Someone turning it off *because they were worried about it* would have believed
|
||||||
|
it was off.
|
||||||
|
|
||||||
|
`apply.sh` now actively reverts: it removes the module first (every injected
|
||||||
|
block is guarded by `if _tc_nested is not None`, so the stock guard is restored
|
||||||
|
even if a later step fails), then strips its appended blocks from the three
|
||||||
|
patched files. `restic.py` also carries a `TRUECLOUD_PATCH` block but belongs to
|
||||||
|
the *providers* module and is deliberately left alone — reverting it would break
|
||||||
|
B2 backups. `install.sh --disable` also tears down any staging tree first, since
|
||||||
|
those bind mounts pin ZFS snapshots that could otherwise never be destroyed.
|
||||||
|
|
||||||
|
Updating **without** the flag was always correct and is unchanged: the nested
|
||||||
|
module is never installed into middleware unless it is explicitly enabled.
|
||||||
|
|
||||||
|
## v0.3.1 — 2026-07-13
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- **Automated releases.** Pushing a `v*` tag runs the full test suite and then
|
||||||
|
cuts a GitHub release whose body is the matching `CHANGELOG.md` section — so
|
||||||
|
release notes have exactly one source of truth, and no second place to go stale.
|
||||||
|
The workflow refuses to publish if the tests fail, if the tag does not match the
|
||||||
|
`VERSION=` declared by every script, or if the CHANGELOG has no section for it.
|
||||||
|
|
||||||
|
- **Version-drift check.** `VERSION=` had silently diverged to three different
|
||||||
|
values across `install.sh`, `uninstall.sh`, `recover.sh`, and `patch/apply.sh`,
|
||||||
|
and nothing noticed. CI now asserts every script agrees with the others and with
|
||||||
|
the newest CHANGELOG entry.
|
||||||
|
|
||||||
|
### Note
|
||||||
|
|
||||||
|
- Releases for `v0.2.0` and `v0.2.1` were backfilled — they had been tagged but
|
||||||
|
never released, so the releases page jumped v0.1.0 → v0.3.0 and hid the fix for
|
||||||
|
the boot race that took every app down.
|
||||||
|
|
||||||
|
## v0.3.0 — 2026-07-13
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- **`snapshot = true` now works on datasets that have child datasets** —
|
||||||
|
**opt-in, off by default** (`install.sh --enable-nested-snapshots` /
|
||||||
|
`--disable-nested-snapshots`). It changes how backups read their source data,
|
||||||
|
so it is never enabled implicitly; with neither flag `install.sh` preserves
|
||||||
|
the existing setting, so a `git pull && bash install.sh` cannot silently flip
|
||||||
|
it. When disabled, `apply.sh` skips the patch entirely and the stock guard
|
||||||
|
remains. `uninstall.sh` tears down any staging mounts and removes the marker.
|
||||||
|
Stock TrueNAS refuses this with *"This option is only available for datasets
|
||||||
|
that have no further nesting"*, which makes the snapshot option unusable for
|
||||||
|
the single most common case on any box running Apps — every app is its own
|
||||||
|
dataset, often with `config`/`pgdata` children of its own. Without it, the
|
||||||
|
backup reads **live** files: databases are captured mid-write, and a busy app
|
||||||
|
rewriting its files can stall a backup indefinitely as restic chases a moving
|
||||||
|
target.
|
||||||
|
|
||||||
|
The stock guard is **correct, and it is not an arbitrary limit.**
|
||||||
|
`plugins/cloud/snapshot.py` already takes a *recursive* ZFS snapshot, but it
|
||||||
|
then points the backup tool at the **parent** dataset's
|
||||||
|
`.zfs/snapshot/<snap>/` directory — and ZFS does not expose child datasets
|
||||||
|
through a parent's snapshot directory:
|
||||||
|
|
||||||
|
```
|
||||||
|
/mnt/Tap/.zfs/snapshot/<snap>/apps/ -> 0 entries (children invisible)
|
||||||
|
/mnt/Tap/apps/lidarr/config/.zfs/snapshot/<snap>/ -> the real data
|
||||||
|
```
|
||||||
|
|
||||||
|
So without the guard the backup tool would walk a near-empty tree, report
|
||||||
|
SUCCESS, and upload almost nothing. iX gate the config rather than ship a
|
||||||
|
backup that lies about succeeding.
|
||||||
|
|
||||||
|
This release implements the missing half. After the (already recursive)
|
||||||
|
snapshot is taken, every descendant dataset's own `.zfs/snapshot/<snap>` is
|
||||||
|
bind-mounted into a **staging tree** mirroring the original layout, and the
|
||||||
|
backup tool is pointed at the staging root — a complete, consistent,
|
||||||
|
point-in-time view of the whole subtree. Only then is the guard relaxed.
|
||||||
|
|
||||||
|
Safety properties, in order of importance:
|
||||||
|
|
||||||
|
- **Staging failure is loud.** If any descendant cannot be staged, the backup
|
||||||
|
fails. A silently-incomplete backup is the exact outcome the stock guard
|
||||||
|
exists to prevent, and it would be worse than not having the feature.
|
||||||
|
- **A post-mount verification pass** asserts every planned target is really a
|
||||||
|
mountpoint and the staging root is non-empty, so this can never regress into
|
||||||
|
the empty-backup failure it is meant to fix.
|
||||||
|
- **The guard is relaxed last.** `apply.sh` installs the traversal, patches
|
||||||
|
`snapshot.py`, then `sync.py`, and only then `crud.py`. A partial failure
|
||||||
|
leaves the guard intact and the option merely unavailable — never
|
||||||
|
"guard removed, traversal missing".
|
||||||
|
- **The patch owns the whole snapshot lifecycle.** `zfs.snapshot.delete`
|
||||||
|
defaults to `recursive=False` and stock `restic_backup()` calls it with no
|
||||||
|
options. Stock gets away with that only because its validation means
|
||||||
|
`recursive` is never True in the field — but enabling nested datasets makes
|
||||||
|
recursive snapshots real, so the parent now has one child snapshot per
|
||||||
|
descendant dataset (160+ on a typical Apps pool). Relying on stock's delete
|
||||||
|
would therefore orphan every child snapshot **on every successful run**.
|
||||||
|
This patch sweeps the parent *and* all children, is idempotent against
|
||||||
|
stock's `finally` winning the race, records the snapshot in a sidecar file
|
||||||
|
(so a middlewared restart mid-backup cannot orphan it), reclaims the tree
|
||||||
|
left by a crashed run, and deletes the tree when staging fails — where
|
||||||
|
sync.py's own `finally` would otherwise delete nothing at all, because its
|
||||||
|
`snapshot` local never gets assigned.
|
||||||
|
- **The dataset list is enumerated *after* the snapshot, never before.** A
|
||||||
|
list read beforehand can miss a dataset created in the gap: the recursive
|
||||||
|
snapshot would capture it but the staging plan would not, silently omitting
|
||||||
|
its data. Read afterwards, an unsnapshotted dataset trips the staging check
|
||||||
|
and fails the run loudly instead.
|
||||||
|
- **Every injected block no-ops** if `_truecloud_nested` is absent.
|
||||||
|
- Datasets that cannot contribute to a file tree (`mountpoint=none|legacy`,
|
||||||
|
unmounted/locked, encrypted-and-locked) are skipped and **reported** —
|
||||||
|
never dropped silently.
|
||||||
|
- Scoped to `cloud_backup` only. Cloud Sync (rclone) shares the same
|
||||||
|
validation mixin but has no staging teardown wired in, so its guard is left
|
||||||
|
in place deliberately.
|
||||||
|
|
||||||
|
Side benefit: the staging root is a **stable** path per task, so restic can
|
||||||
|
find its parent snapshot between runs. Stock's
|
||||||
|
`.zfs/snapshot/<name>-<timestamp>/` path changes every run, which defeats
|
||||||
|
restic's parent detection and forces a full re-scan each time.
|
||||||
|
|
||||||
|
- **CI** (GitHub Actions): shellcheck + `bash -n` on every script, ruff, and
|
||||||
|
pytest on Python 3.11/3.12/3.13. Includes tests that `compile()` the
|
||||||
|
`*_BLOCK` strings — they are Python source appended to live middlewared
|
||||||
|
modules, so a syntax error there would break the box at boot, and nothing
|
||||||
|
previously checked them.
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- **The patch is now two independent modules, and each retires on its own.**
|
||||||
|
Previously the native-support check looked only for native B2 restic support
|
||||||
|
and, on finding it, set the kill switch and disabled *everything*. With a
|
||||||
|
second capability in the patch that would silently take a still-needed module
|
||||||
|
down with the superseded one — TrueNAS is likely to ship one of these long
|
||||||
|
before the other.
|
||||||
|
|
||||||
|
`apply.sh` now detects each separately (`providers`: does `B2RcloneRemote`
|
||||||
|
carry a real `get_restic_config()`; `nested`: is the *"no further nesting"*
|
||||||
|
validation still in `plugins/cloud/crud.py`), skips just the superseded one,
|
||||||
|
and only sets the kill switch once **both** are done. The UI patch belongs to
|
||||||
|
`providers` and is skipped with it. The deferred middlewared restart now fires
|
||||||
|
when *any* still-needed module landed — keying it off `providers` alone would
|
||||||
|
have left a freshly-patched `nested` module on disk and never loaded on a
|
||||||
|
native-B2 box. `hook_status.json` reports each module with an `active` flag and
|
||||||
|
a reason.
|
||||||
|
|
||||||
|
- README rewritten to be less alarmist: dropped the warning boxes and the
|
||||||
|
disclaimer's fear-bulleting in favour of plain statements, and documented the
|
||||||
|
two-module design. The one caveat kept as a plain sentence: the `mount --bind`
|
||||||
|
staging step has not yet been exercised by a live backup run.
|
||||||
|
|
||||||
|
- Version strings in `install.sh`, `uninstall.sh`, and `recover.sh` were stale
|
||||||
|
at `0.0.4`; all scripts now report the same version.
|
||||||
|
- `patch_ui.py`: replaced a `try`/`except`/`pass` with `contextlib.suppress`
|
||||||
|
(no behaviour change; satisfies the new lint gate).
|
||||||
|
|
||||||
|
### Removed
|
||||||
|
|
||||||
|
- `patch/__pycache__/create_task.cpython-314.pyc` was committed to the
|
||||||
|
repository; it is now untracked and `__pycache__/` is gitignored.
|
||||||
|
|
||||||
|
### Fixed (post-merge audit)
|
||||||
|
|
||||||
|
- **`create_task.py verify` failed on a default install.** `hook_status.json`
|
||||||
|
emitted a per-file entry for the nested module with `ok: false` whenever the
|
||||||
|
feature was switched off — which is the default — so `verify` printed `[FAIL]`
|
||||||
|
and exited 1 right after the README told users to run it. Status is now
|
||||||
|
reported per *module* with an `active` flag, and `verify` renders an inactive
|
||||||
|
module as `[SKIP]` rather than a failure.
|
||||||
|
- **A partial apply suppressed the middlewared restart.** The exit code
|
||||||
|
conflated "nothing applied" with "one module applied, one failed", so a failing
|
||||||
|
providers patch would prevent the restart that a freshly-applied nested patch
|
||||||
|
needs — leaving it on disk and never loaded. Exit 2 now means partial, and the
|
||||||
|
restart still fires.
|
||||||
|
- **The native-nested probe could never fire.** It scanned `crud.py` for the
|
||||||
|
guard message, but our own injected block *quotes* that message, so once
|
||||||
|
applied the probe would always conclude the guard was still present. It now
|
||||||
|
reads only the stock portion of the file.
|
||||||
|
- `recover.sh` did not unmount staging trees, so an emergency recovery left bind
|
||||||
|
mounts pinning ZFS snapshots that could then never be destroyed.
|
||||||
|
- `uninstall.sh` deleted sidecar files without reading them. A sidecar is the
|
||||||
|
only record that an interrupted run's snapshot tree is still on disk; both
|
||||||
|
scripts now name the snapshot (`zfs destroy -r ...`) before clearing it.
|
||||||
|
|
||||||
|
- **The native-nested probe could never detect the guard, silently disabling the
|
||||||
|
whole module.** Stock splits the message across adjacent string literals:
|
||||||
|
|
||||||
|
```python
|
||||||
|
verrors.add(f"{name}.snapshot", "This option is only available for datasets that have no further "
|
||||||
|
"nesting")
|
||||||
|
```
|
||||||
|
|
||||||
|
Python concatenates those at runtime — so the *errmsg* is contiguous and the
|
||||||
|
runtime filter works — but the **source never contains the whole phrase**. The
|
||||||
|
probe's substring search found nothing, concluded iX had removed the guard, and
|
||||||
|
skipped the nested module as "already native". `apply.log` would report
|
||||||
|
*"TrueNAS now handles nesting natively"* and the feature would never work.
|
||||||
|
It fails safe (the stock guard stays, so no data is at risk) but the module was
|
||||||
|
100% dead. The probe now strips whitespace and quotes before matching, which is
|
||||||
|
robust to any wrapping style. Caught only by running the probe against real
|
||||||
|
middlewared; there is now a regression test that executes apply.sh's own probe
|
||||||
|
code against the real wrapped source.
|
||||||
|
|
||||||
|
### Changed (production audit)
|
||||||
|
|
||||||
|
- **`delete_snapshot_tree` now uses a single recursive delete.** It previously
|
||||||
|
removed the parent and each child snapshot one at a time — 252 sequential
|
||||||
|
middleware calls on a real pool. That is slow, but the real problem is that it
|
||||||
|
is **not atomic**: a run killed part-way through the sweep leaves exactly the
|
||||||
|
orphaned snapshots the function exists to prevent. It now issues one
|
||||||
|
`zfs.snapshot.delete(..., {"recursive": True})` and falls back to the
|
||||||
|
name-by-name sweep only when that fails (e.g. stock's `finally` already removed
|
||||||
|
the parent, which leaves the children behind).
|
||||||
|
|
||||||
|
### Refactored
|
||||||
|
|
||||||
|
- Staging teardown had been copy-pasted into `uninstall.sh` and `recover.sh` —
|
||||||
|
two untested shell copies of the fiddly depth-ordering and lazy-umount logic.
|
||||||
|
Both now call `python3 patch/truecloud_nested.py cleanup`, so there is one
|
||||||
|
implementation and it is the one under test.
|
||||||
|
- Dropped the in-memory `ACTIVE` dict. The sidecar file was already the source of
|
||||||
|
truth; a second in-process record could only desync — and it is the
|
||||||
|
middlewared-restart case (which empties it) that must not orphan a snapshot
|
||||||
|
tree. One record, on disk, or none.
|
||||||
|
|
||||||
|
### Validated in production
|
||||||
|
|
||||||
|
An unattended scheduled backup of a live 252-dataset pool (`/mnt/Tap`, TrueNAS
|
||||||
|
25.10) ran through the staging tree end to end:
|
||||||
|
|
||||||
|
- 252 datasets recursively snapshotted, 173 bind mounts built and verified
|
||||||
|
- completed in **18m14s**, `SUCCESS` — the same task previously stalled at 74%
|
||||||
|
for over 12 hours reading live files
|
||||||
|
- **zero** orphaned ZFS snapshots and **zero** stale mounts afterwards, which is
|
||||||
|
the failure mode that would otherwise have accumulated 251 snapshots per run
|
||||||
|
|
||||||
|
### Known issues
|
||||||
|
|
||||||
|
- Stock `restic_backup()` deletes the ZFS snapshot in its own `finally`, which
|
||||||
|
fails with `EBUSY` while the staging bind mounts pin it. It logs one benign
|
||||||
|
`Error deleting snapshot ...` warning per run; the patch then unmounts and
|
||||||
|
deletes the snapshot for real. The warning is expected and harmless.
|
||||||
|
|
||||||
|
## v0.2.1 — 2026-07-09
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Deferred restart raced the rest of boot, leaving all apps and dashboard
|
||||||
|
stats down.** The `truecloud-mw-restart` unit introduced in v0.0.4 relied
|
||||||
|
on systemd ordering (`After=multi-user.target`, `After=ix-postinit.service`),
|
||||||
|
which cannot see middlewared's *internal* boot work. Observed on 25.10.4:
|
||||||
|
the restart fired two seconds into `ix-reporting.service`'s
|
||||||
|
`midclt call reporting.start_service` and before the docker/apps startup
|
||||||
|
task (created on middlewared's system-ready event) had run. Both were
|
||||||
|
killed, and nothing retries them until the next boot — every app stayed
|
||||||
|
down (`docker.status` FAILED, the apps dataset never mounted), netdata
|
||||||
|
never started (no dashboard hardware stats), and the SMB middleware
|
||||||
|
backend was left uninitialized.
|
||||||
|
|
||||||
|
The transient unit now runs `patch/wait_restart.sh` instead of restarting
|
||||||
|
directly: it waits for the systemd boot job queue to drain
|
||||||
|
(`systemctl is-system-running --wait`, covering in-flight `ix-*` oneshots
|
||||||
|
such as ix-reporting), then polls `midclt call docker.status` until the
|
||||||
|
docker state machine leaves its transitional states, then allows a short
|
||||||
|
grace period for middleware-internal tasks with no queryable state before
|
||||||
|
issuing `systemctl try-restart middlewared`. The unit no longer sets
|
||||||
|
`Type=oneshot` — a oneshot's start job stays in the very queue the script
|
||||||
|
waits on and would deadlock on itself. All waits are bounded and fail
|
||||||
|
open: worst case the restart still happens, just later.
|
||||||
|
|
||||||
|
Recovery on a boot that already hit this (without rebooting):
|
||||||
|
`midclt call reporting.start_service` and
|
||||||
|
`midclt call docker.state.start_service true`.
|
||||||
|
|
||||||
## v0.2.0 — 2026-07-08
|
## v0.2.0 — 2026-07-08
|
||||||
|
|
||||||
### Changed
|
### Changed
|
||||||
|
|||||||
@@ -1,7 +1,10 @@
|
|||||||
# truenas-truecloud-patch
|
# truenas-truecloud-patch
|
||||||
|
|
||||||
Extends TrueNAS SCALE's **TrueCloud Backup** feature to work with S3-compatible
|
Extends TrueNAS SCALE's **TrueCloud Backup** feature to:
|
||||||
providers and native Backblaze B2, instead of Storj only.
|
|
||||||
|
- work with S3-compatible providers and native Backblaze B2, instead of Storj only;
|
||||||
|
- take **consistent snapshots of datasets that have child datasets** — which is
|
||||||
|
every box running Apps (see [Nested-dataset snapshots](#nested-dataset-snapshots)).
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -24,31 +27,41 @@ costs a fraction of the new Storj price.
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## ⚠ Disclaimer — please read before installing
|
## Before you install
|
||||||
|
|
||||||
**This project is unofficial, unsupported, and not affiliated with iXsystems
|
This project is unofficial and not affiliated with iXsystems. A few things worth
|
||||||
or the TrueNAS project in any way.**
|
knowing:
|
||||||
|
|
||||||
By installing this patch you accept the following:
|
- It targets **internal middleware APIs** with no stability contract, so a
|
||||||
|
TrueNAS update can break it. Every patch is fail-safe: if it can't apply,
|
||||||
|
middlewared starts normally and the reason is logged to `apply.log`. Check the
|
||||||
|
log after an update.
|
||||||
|
- If you file a TrueNAS bug report, **remove the patch first** and reproduce on a
|
||||||
|
stock system.
|
||||||
|
- **Test your restores.** True of any backup, but it matters more here — see
|
||||||
|
[Verifying it works](#verifying-it-works).
|
||||||
|
- Provided as-is, no warranty. See LICENSE.
|
||||||
|
|
||||||
- **Unsupported configuration.** TrueNAS support staff are not obligated to
|
The patch is two independent modules — **providers** (B2/S3) and **nested**
|
||||||
help with any issue on a system running this patch. If you file a bug report,
|
(snapshots on nested datasets) — and each retires on its own once TrueNAS ships
|
||||||
remove the patch first and reproduce the issue on an unmodified system.
|
that capability natively. See [Native support](#if-truenas-adds-native-support).
|
||||||
|
|
||||||
- **May break on TrueNAS updates.** The patch targets internal middleware APIs
|
## Development
|
||||||
that are not part of any public contract. They can change at any time. When
|
|
||||||
they do, the patch silently degrades to Storj-only behaviour rather than
|
|
||||||
breaking TrueNAS — but you should check the log after each update.
|
|
||||||
|
|
||||||
- **Your backups are your responsibility.** Verify that your backup jobs
|
Parts of this project were written with AI assistance (Claude). All of it is
|
||||||
complete successfully and that restores work before relying on them for
|
reviewed and tested before release; the test suite and CI exist in large part to
|
||||||
disaster recovery.
|
make that review meaningful. Bugs are mine.
|
||||||
|
|
||||||
- **No warranty.** This software is provided as-is. See the LICENSE file.
|
```bash
|
||||||
|
pip install pytest ruff
|
||||||
|
ruff check patch tests
|
||||||
|
pytest tests
|
||||||
|
```
|
||||||
|
|
||||||
If TrueNAS adds native B2 or S3 support to TrueCloud Backup, the patch
|
CI runs shellcheck, `bash -n`, ruff, and pytest on Python 3.11–3.13. The tests
|
||||||
detects it at boot, disables itself, and tells you to run `uninstall.sh` —
|
include a pass that `compile()`s the `*_BLOCK` strings in `patch/apply.sh` —
|
||||||
see [Native support](#if-truenas-adds-native-support) below.
|
those are Python source appended into live `middlewared` modules, so a syntax
|
||||||
|
error there would break the box at boot.
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -66,20 +79,181 @@ directly.
|
|||||||
|
|
||||||
PREINIT scripts are executed *by* middlewared, which by then has already
|
PREINIT scripts are executed *by* middlewared, which by then has already
|
||||||
imported the stock modules — so after patching, `apply.sh` schedules a single
|
imported the stock modules — so after patching, `apply.sh` schedules a single
|
||||||
detached middlewared restart (transient systemd unit `truecloud-mw-restart`,
|
detached middlewared restart (transient systemd unit `truecloud-mw-restart`
|
||||||
ordered after `multi-user.target`) that loads the patched modules once boot
|
running `patch/wait_restart.sh`) that loads the patched modules once boot has
|
||||||
completes. Expect one middlewared restart shortly after every boot; the UI
|
*actually* settled: the script waits for the systemd boot job queue to drain
|
||||||
|
and for the docker/apps state machine to reach a terminal state before
|
||||||
|
restarting. Expect one middlewared restart shortly after every boot; the UI
|
||||||
and API are briefly unavailable while it happens, and running services are
|
and API are briefly unavailable while it happens, and running services are
|
||||||
not affected.
|
not affected.
|
||||||
|
|
||||||
| Layer | What changes | Technique |
|
| Module | What changes | Technique |
|
||||||
|---|---|---|
|
|---|---|---|
|
||||||
| **Backend** | `B2RcloneRemote` gains `get_restic_config()` — skipped automatically if TrueNAS already provides one on the class. `restic.py` URL builder is fixed: strips the stray leading slash and converts the slash separator to a colon (`b2:bucket:path`), which is the format restic 0.16.x expects. URL wrapper is a no-op if the URL is already correctly formed. | File patch applied inside the overlayfs upper layer |
|
| **providers** | `B2RcloneRemote` gains `get_restic_config()` — skipped automatically if TrueNAS already provides one on the class. `restic.py` URL builder is fixed: strips the stray leading slash and converts the slash separator to a colon (`b2:bucket:path`), which is the format restic 0.16.x expects. URL wrapper is a no-op if the URL is already correctly formed. | File patch applied inside the overlayfs upper layer |
|
||||||
| **UI** | The Angular bundle's `filterByProviders` binding is widened from `["STORJ_IX"]` to `["STORJ_IX","S3","B2"]` | In-place text replacement in the compiled JS chunk; original is backed up before patching |
|
| **providers** (UI) | The Angular bundle's `filterByProviders` binding is widened from `["STORJ_IX"]` to `["STORJ_IX","S3","B2"]` | In-place text replacement in the compiled JS chunk; original is backed up before patching |
|
||||||
|
| **nested** (opt-in) | `_truecloud_nested.py` is installed into `plugins/cloud/`, and `plugins/cloud/{snapshot,crud}.py` + `plugins/cloud_backup/sync.py` are patched so `snapshot = true` works on a dataset that has child datasets. See [below](#nested-dataset-snapshots). | New module + file patches inside the overlayfs upper layer |
|
||||||
|
|
||||||
Both changes are **fail-safe**: if a patch cannot be applied (e.g. TrueNAS
|
All changes are **fail-safe**: if a patch cannot be applied (e.g. TrueNAS
|
||||||
restructured the relevant code), middlewared starts normally with Storj-only
|
restructured the relevant code), middlewared starts normally, the affected module
|
||||||
support and the reason is logged to `apply.log` in your repo root.
|
is simply inactive, and the reason is logged to `apply.log` in your repo root. The
|
||||||
|
two modules are independent — one failing or going native does not disable the
|
||||||
|
other.
|
||||||
|
|
||||||
|
## Nested-dataset snapshots
|
||||||
|
|
||||||
|
**Opt-in, off by default.** It changes how backups read their source data, so it
|
||||||
|
is never enabled implicitly:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
bash install.sh --enable-nested-snapshots
|
||||||
|
bash install.sh --disable-nested-snapshots
|
||||||
|
```
|
||||||
|
|
||||||
|
With neither flag `install.sh` leaves the setting alone, so `git pull && bash
|
||||||
|
install.sh` won't flip it. The providers module is unaffected either way.
|
||||||
|
|
||||||
|
Validated end to end on a live 252-dataset pool: an unattended scheduled backup
|
||||||
|
of `/mnt/Tap` built a 173-mount staging tree, completed in **18m14s**, and left
|
||||||
|
**zero** orphaned snapshots and **zero** stale mounts behind. The same backup
|
||||||
|
previously stalled at 74% for over 12 hours reading live files.
|
||||||
|
|
||||||
|
Still: verify your own first run actually contains child-dataset data before you
|
||||||
|
rely on it — see [Verifying it works](#verifying-it-works). That advice is not
|
||||||
|
boilerplate; it is the specific thing this feature exists to make true.
|
||||||
|
|
||||||
|
TrueCloud Backup's **Take Snapshot** option makes restic read from a frozen ZFS
|
||||||
|
snapshot instead of live files. Without it the backup reads data *while apps are
|
||||||
|
writing to it* — databases get captured mid-write, and an app that rewrites its
|
||||||
|
files continuously can stall a backup indefinitely as restic chases a moving
|
||||||
|
target.
|
||||||
|
|
||||||
|
Stock TrueNAS refuses to enable it on most real-world paths:
|
||||||
|
|
||||||
|
```
|
||||||
|
[EINVAL] cloud_backup_update.snapshot:
|
||||||
|
This option is only available for datasets that have no further nesting
|
||||||
|
```
|
||||||
|
|
||||||
|
That rules out **any pool running Apps** — every app is its own dataset, usually
|
||||||
|
with `config`/`pgdata` children of its own. On a typical box that is 100+ nested
|
||||||
|
datasets, so the feature is effectively unusable exactly where it matters most.
|
||||||
|
|
||||||
|
### Why stock refuses
|
||||||
|
|
||||||
|
The guard is **correct**. `plugins/cloud/snapshot.py` already takes a *recursive*
|
||||||
|
ZFS snapshot — but it then points restic at the **parent** dataset's
|
||||||
|
`.zfs/snapshot/<snap>/` directory, and ZFS does not expose child datasets
|
||||||
|
through a parent's snapshot directory:
|
||||||
|
|
||||||
|
```
|
||||||
|
/mnt/Tap/.zfs/snapshot/<snap>/apps/ -> 0 entries (children invisible)
|
||||||
|
/mnt/Tap/apps/lidarr/config/.zfs/snapshot/<snap>/ -> the real data
|
||||||
|
```
|
||||||
|
|
||||||
|
So if you just remove the validation, restic walks a near-empty tree, reports
|
||||||
|
SUCCESS, and uploads almost nothing — a green backup job protecting no data. iX
|
||||||
|
gate the config rather than ship a backup that lies about succeeding.
|
||||||
|
|
||||||
|
That is worth spelling out, because deleting those four lines in
|
||||||
|
`plugins/cloud/crud.py` is the obvious "fix" and it is the wrong one. The guard
|
||||||
|
is load-bearing: it has to be *replaced* with a working traversal, not removed.
|
||||||
|
|
||||||
|
### What this patch does instead
|
||||||
|
|
||||||
|
After the (already recursive) snapshot is taken, every descendant dataset's own
|
||||||
|
`.zfs/snapshot/<snap>` is bind-mounted into a **staging tree** that mirrors the
|
||||||
|
original layout, and restic is pointed at the staging root — a complete,
|
||||||
|
consistent, point-in-time view of the whole subtree. Only then is the guard
|
||||||
|
relaxed.
|
||||||
|
|
||||||
|
Safety properties, in order of importance:
|
||||||
|
|
||||||
|
- **Staging failure is loud.** If any descendant cannot be staged, the backup
|
||||||
|
*fails*. A silently-incomplete backup is precisely what the stock guard exists
|
||||||
|
to prevent, and it would be worse than not having the feature at all.
|
||||||
|
- **Post-mount verification** asserts every planned target really is a mountpoint
|
||||||
|
and the staging root is non-empty — so this cannot regress into the empty
|
||||||
|
backup it exists to fix.
|
||||||
|
- **The guard is relaxed last.** `apply.sh` installs the traversal, patches
|
||||||
|
`snapshot.py`, then `sync.py`, and only then `crud.py`. Any partial failure
|
||||||
|
leaves the guard intact and the option merely unavailable — never
|
||||||
|
"guard removed, traversal missing".
|
||||||
|
- Datasets that cannot contribute to a file tree (`mountpoint=none|legacy`,
|
||||||
|
unmounted, locked/encrypted) are skipped and **reported to the log** — never
|
||||||
|
dropped silently.
|
||||||
|
- Scoped to **cloud_backup only**. Cloud Sync (rclone) shares the same
|
||||||
|
validation mixin but has no staging teardown wired in, so its guard is
|
||||||
|
deliberately left in place.
|
||||||
|
|
||||||
|
Side benefit: the staging root is a **stable path per task**, so restic can find
|
||||||
|
its parent snapshot between runs. Stock's `.zfs/snapshot/<name>-<timestamp>/`
|
||||||
|
path changes every run, defeating restic's parent detection and forcing a full
|
||||||
|
re-scan each time.
|
||||||
|
|
||||||
|
### Snapshot lifecycle
|
||||||
|
|
||||||
|
`zfs.snapshot.delete` defaults to **`recursive=False`**, and stock
|
||||||
|
`restic_backup()` calls it with no options. Stock is safe only because its
|
||||||
|
validation means a *recursive* snapshot never actually happens in the field.
|
||||||
|
Enabling nested datasets makes them real: on a 250-dataset pool,
|
||||||
|
`zfs snapshot -r` creates **250 snapshots**, and stock's delete removes only the
|
||||||
|
parent — orphaning **249 on every successful run** (measured, not theorised).
|
||||||
|
|
||||||
|
So the patch owns the whole lifecycle:
|
||||||
|
|
||||||
|
- **Sweeps the parent and every child**, and is idempotent against stock's
|
||||||
|
`finally` winning the race once the mounts are released.
|
||||||
|
- **Records the snapshot in a sidecar file before mounting anything**, so a
|
||||||
|
middlewared restart mid-backup cannot orphan the tree (this patch *schedules*
|
||||||
|
a restart at boot, so that is not hypothetical).
|
||||||
|
- **Reclaims the tree left by a crashed run** instead of overwriting the record.
|
||||||
|
- **Deletes the tree when staging fails** — sync.py's own `finally` deletes
|
||||||
|
*nothing* in that case, because its `snapshot` local never gets assigned.
|
||||||
|
- **Enumerates datasets *after* the snapshot, never before.** A list read
|
||||||
|
beforehand can miss a dataset created in the gap, which the recursive snapshot
|
||||||
|
*would* capture but the staging plan would not — a silent omission.
|
||||||
|
|
||||||
|
**Expected log noise:** stock's delete fails with `EBUSY` while the staging
|
||||||
|
mounts pin the snapshot. You will see one benign `Error deleting snapshot ...`
|
||||||
|
warning per run; the patch then unmounts and deletes the tree for real.
|
||||||
|
|
||||||
|
### Verifying it works
|
||||||
|
|
||||||
|
This feature exists because a backup can report SUCCESS while containing
|
||||||
|
nothing, so check the contents rather than the exit status:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# 1. Does the restic snapshot actually contain child-dataset data?
|
||||||
|
# Pick a path that lives in a CHILD dataset (e.g. an app's config).
|
||||||
|
midclt call cloud_backup.list_snapshots <task_id> | head
|
||||||
|
|
||||||
|
# 2. List a child-dataset path inside the newest restic snapshot.
|
||||||
|
# If this is empty, the staging tree did not work and you are backing up NOTHING.
|
||||||
|
midclt call cloud_backup.list_snapshot_directory <task_id> "<snapshot_id>" "/apps/lidarr/config"
|
||||||
|
```
|
||||||
|
|
||||||
|
You should see the app's real files (`lidarr.db`, `config.xml`, …). An empty
|
||||||
|
listing means the child datasets were not staged; disable the feature and open an
|
||||||
|
issue.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# 3. No snapshots may be left behind after a run.
|
||||||
|
zfs list -t snapshot -r <pool> | grep -c cloud_backup- # expect 0 between runs
|
||||||
|
|
||||||
|
# 4. No staging mounts may be left behind.
|
||||||
|
mount | grep truecloud-nested # expect no output
|
||||||
|
```
|
||||||
|
|
||||||
|
### Troubleshooting
|
||||||
|
|
||||||
|
| Symptom | Cause |
|
||||||
|
|---|---|
|
||||||
|
| `This option is only available for datasets that have no further nesting` | Feature not enabled. Run `install.sh --enable-nested-snapshots`, then restart middlewared. |
|
||||||
|
| Backup fails: `dataset '…' has no snapshot '…'; refusing to back up an incomplete tree` | Working as designed — a descendant dataset was not covered by the snapshot. The backup is refused rather than silently omitting that data. |
|
||||||
|
| Backup fails: `snapshot '…' cannot be read (Permission denied)` | The snapshot exists but is unreadable. Middleware runs as root, so this indicates a real permissions problem, not a missing snapshot. |
|
||||||
|
| `cloud_backup-*` snapshots accumulating | The sweep is not running. Check `apply.log` for the nested patch applying, and confirm `sync.py` carries the `TRUECLOUD_PATCH` block. |
|
||||||
|
| Web UI blank after a patch | A bad pattern unbalanced the bundle. `apply.sh` now refuses to write in that case, but if you hit it on an older version: restore `chunk-*.js.pre-truecloud-patch` over the live chunk, then re-run `install.sh`. (`MARKER` makes an already-patched file skip, so the patch cannot heal a corrupted bundle by itself.) |
|
||||||
|
| Stale mounts under `/run/truecloud-nested` | A crashed run. The next backup tears them down. To clear them now: `python3 patch/truecloud_nested.py cleanup` (also run by `uninstall.sh` and `recover.sh`). It names any ZFS snapshot an interrupted run left pinned. |
|
||||||
|
|
||||||
## Supported providers after patching
|
## Supported providers after patching
|
||||||
|
|
||||||
@@ -113,10 +287,15 @@ Two different things must survive two different events:
|
|||||||
imported the stock modules in step 1 and never re-imports, so the on-disk
|
imported the stock modules in step 1 and never re-imports, so the on-disk
|
||||||
patch alone is not enough. `apply.sh` detects it was invoked by middlewared
|
patch alone is not enough. `apply.sh` detects it was invoked by middlewared
|
||||||
and creates a transient systemd unit (`truecloud-mw-restart`, via
|
and creates a transient systemd unit (`truecloud-mw-restart`, via
|
||||||
`systemd-run --no-block`, ordered after `multi-user.target`) — detached and
|
`systemd-run --no-block`) running `patch/wait_restart.sh` — detached so it
|
||||||
deferred so it cannot disrupt the remainder of the boot sequence.
|
cannot disrupt the remainder of the boot sequence.
|
||||||
5. **Once boot completes, middlewared restarts once** and imports the patched
|
5. **Once boot has settled, middlewared restarts once** and imports the
|
||||||
modules from the overlay. S3/B2 backup support is now active until the next
|
patched modules from the overlay. `wait_restart.sh` holds the restart until
|
||||||
|
the systemd boot job queue has drained (so in-flight `ix-*` units like
|
||||||
|
`ix-reporting` finish first) *and* middlewared's docker/apps startup has
|
||||||
|
reached a terminal state — plain unit ordering cannot see either, and
|
||||||
|
restarting middlewared while they run kills apps and dashboard reporting
|
||||||
|
for the whole boot. S3/B2 backup support is then active until the next
|
||||||
reboot, when the cycle repeats.
|
reboot, when the cycle repeats.
|
||||||
|
|
||||||
What you will observe: one middlewared restart shortly after every boot (a
|
What you will observe: one middlewared restart shortly after every boot (a
|
||||||
@@ -232,25 +411,35 @@ and restores the original UI bundle from backup.
|
|||||||
|
|
||||||
## If TrueNAS adds native support
|
## If TrueNAS adds native support
|
||||||
|
|
||||||
`apply.sh` checks at every boot whether TrueNAS has shipped native B2 restic
|
The patch is **two independent modules**, and each retires on its own — TrueNAS
|
||||||
support (by inspecting `B2RcloneRemote.__dict__`). If it has:
|
is likely to ship one of these natively long before the other, and a module
|
||||||
|
going native must not take the other one down with it.
|
||||||
|
|
||||||
1. The kill switch (`disabled` file) is set — no patching on any future boot.
|
| Module | What it does | Detected as native when |
|
||||||
2. Any active overlays are unmounted immediately.
|
|---|---|---|
|
||||||
3. The following message is written to `apply.log`:
|
| **providers** | B2/S3 credentials for TrueCloud Backup (`b2.py`, `restic.py`, UI dropdown) | `B2RcloneRemote` carries a real `get_restic_config()` |
|
||||||
|
| **nested** | Snapshots on datasets with child datasets (`plugins/cloud/*`) | the *"no further nesting"* validation is gone from `plugins/cloud/crud.py` |
|
||||||
|
|
||||||
```
|
At every boot `apply.sh` checks both:
|
||||||
NOTICE: TrueNAS now provides native B2 restic support — truecloud-patch is no longer needed.
|
|
||||||
NOTICE: Setting kill switch; patching will be skipped on all future boots.
|
- **One module goes native** → that module is skipped and logged; the other keeps
|
||||||
NOTICE: Run the following to fully remove the patch:
|
working, and the patch stays installed.
|
||||||
NOTICE: bash /mnt/tank/truenas-truecloud-patch/uninstall.sh
|
- **Both are done** (native, or nested was never enabled) → the kill switch
|
||||||
```
|
(`disabled` file) is set, overlays are unmounted, and `apply.log` tells you to
|
||||||
|
run `uninstall.sh`.
|
||||||
|
|
||||||
|
So on a box using only the provider patch, native B2 support retires the whole
|
||||||
|
thing as before. On a box that also uses nested snapshots, native B2 support
|
||||||
|
retires *just* that half.
|
||||||
|
|
||||||
Check the log after any TrueNAS update:
|
Check the log after any TrueNAS update:
|
||||||
```bash
|
```bash
|
||||||
cat /mnt/tank/truenas-truecloud-patch/apply.log | tail -20
|
tail -20 /mnt/tank/truenas-truecloud-patch/apply.log
|
||||||
```
|
```
|
||||||
|
|
||||||
|
`hook_status.json` reports each module separately (`module.providers`,
|
||||||
|
`module.nested_snapshots`) with an `active` flag and a reason.
|
||||||
|
|
||||||
**Scenarios where the auto-detect may not fire** (manual check needed):
|
**Scenarios where the auto-detect may not fire** (manual check needed):
|
||||||
|
|
||||||
| Scenario | What happens | Action |
|
| Scenario | What happens | Action |
|
||||||
@@ -355,19 +544,28 @@ patch re-runs at next reboot, or you run
|
|||||||
python3 /mnt/tank/truenas-truecloud-patch/patch/create_task.py verify
|
python3 /mnt/tank/truenas-truecloud-patch/patch/create_task.py verify
|
||||||
```
|
```
|
||||||
|
|
||||||
If one or more entries show `[FAIL]`:
|
`verify` reports one line per module:
|
||||||
|
|
||||||
|
| Label | Meaning |
|
||||||
|
|---|---|
|
||||||
|
| `[OK ]` | Module is active and applied. |
|
||||||
|
| `[SKIP]` | Module is inactive — either TrueNAS now does it natively, or it is opt-in and switched off. **Not a failure.** `nested_snapshots` shows SKIP on a default install. |
|
||||||
|
| `[FAIL]` | Module is needed but did not apply. |
|
||||||
|
|
||||||
|
If a module shows `[FAIL]`:
|
||||||
|
|
||||||
1. **Check the apply log** for errors during the last boot:
|
1. **Check the apply log** for errors during the last boot:
|
||||||
```bash
|
```bash
|
||||||
cat /mnt/tank/truenas-truecloud-patch/apply.log | tail -40
|
tail -40 /mnt/tank/truenas-truecloud-patch/apply.log
|
||||||
```
|
```
|
||||||
2. **Check middlewared's own log** for Python tracebacks:
|
2. **Check middlewared's own log** for Python tracebacks:
|
||||||
```bash
|
```bash
|
||||||
grep -i "truecloud\|traceback\|error" /var/log/middlewared.log 2>/dev/null | tail -30
|
grep -i "truecloud\|traceback\|error" /var/log/middlewared.log 2>/dev/null | tail -30
|
||||||
journalctl -u middlewared -n 50
|
journalctl -u middlewared -n 50
|
||||||
```
|
```
|
||||||
3. **A FAIL is non-fatal.** middlewared runs normally; the affected provider
|
3. **A FAIL is non-fatal.** middlewared runs normally and the other module is
|
||||||
falls back to Storj-only. Your existing backups are not at risk.
|
unaffected; the failed one is simply inactive. Existing backups are not at
|
||||||
|
risk.
|
||||||
4. **If the detail says the module doesn't exist**, a TrueNAS update renamed
|
4. **If the detail says the module doesn't exist**, a TrueNAS update renamed
|
||||||
or restructured the internal API.
|
or restructured the internal API.
|
||||||
[Open an issue](https://github.com/sudolulo/truenas-truecloud-patch/issues)
|
[Open an issue](https://github.com/sudolulo/truenas-truecloud-patch/issues)
|
||||||
|
|||||||
+83
-1
@@ -18,11 +18,50 @@
|
|||||||
|
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
VERSION="0.0.4"
|
VERSION="0.3.2"
|
||||||
|
|
||||||
# The directory containing install.sh is the permanent install location.
|
# The directory containing install.sh is the permanent install location.
|
||||||
PATCH_DIR="$(cd "$(dirname "$0")" && pwd)"
|
PATCH_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||||
_HOOK_COMMENT='TrueCloud provider patch (S3/B2)'
|
_HOOK_COMMENT='TrueCloud provider patch (S3/B2)'
|
||||||
|
_NESTED_MARKER="$PATCH_DIR/nested_snapshots_enabled"
|
||||||
|
|
||||||
|
# ── Options ───────────────────────────────────────────────────────────────────
|
||||||
|
# Nested-dataset snapshot support is OPT-IN and off by default. It changes how
|
||||||
|
# backups read their source data, so an unattended re-run (e.g. after a
|
||||||
|
# `git pull`) must never flip it on or off by itself: with neither flag given,
|
||||||
|
# whatever was chosen previously is preserved.
|
||||||
|
_nested_choice=""
|
||||||
|
|
||||||
|
usage() {
|
||||||
|
cat <<USAGE
|
||||||
|
Usage: bash install.sh [options]
|
||||||
|
|
||||||
|
Options:
|
||||||
|
--enable-nested-snapshots Allow the "Take Snapshot" option on datasets that
|
||||||
|
have child datasets (every pool running Apps).
|
||||||
|
Stock TrueNAS refuses this; see README. Off by
|
||||||
|
default because it changes how backups read data.
|
||||||
|
--disable-nested-snapshots Turn it back off; the stock guard is restored.
|
||||||
|
-h, --help Show this help.
|
||||||
|
|
||||||
|
With neither flag, the current setting is left unchanged.
|
||||||
|
USAGE
|
||||||
|
}
|
||||||
|
|
||||||
|
while [ $# -gt 0 ]; do
|
||||||
|
case "$1" in
|
||||||
|
--enable-nested-snapshots) _nested_choice="on" ;;
|
||||||
|
--disable-nested-snapshots) _nested_choice="off" ;;
|
||||||
|
-h|--help) usage; exit 0 ;;
|
||||||
|
*)
|
||||||
|
echo "ERROR: unknown option: $1" >&2
|
||||||
|
echo "" >&2
|
||||||
|
usage >&2
|
||||||
|
exit 1
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
shift
|
||||||
|
done
|
||||||
|
|
||||||
if [ ! -f "$PATCH_DIR/patch/apply.sh" ]; then
|
if [ ! -f "$PATCH_DIR/patch/apply.sh" ]; then
|
||||||
echo "ERROR: patch files not found at $PATCH_DIR/patch/" >&2
|
echo "ERROR: patch files not found at $PATCH_DIR/patch/" >&2
|
||||||
@@ -104,6 +143,49 @@ if [ -f "$PATCH_DIR/disabled" ]; then
|
|||||||
echo ""
|
echo ""
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
# ── Nested-dataset snapshot support (opt-in) ──────────────────────────────────
|
||||||
|
|
||||||
|
case "$_nested_choice" in
|
||||||
|
on)
|
||||||
|
touch "$_NESTED_MARKER"
|
||||||
|
echo "Nested-dataset snapshots: ENABLED"
|
||||||
|
echo " The \"Take Snapshot\" option will be allowed on datasets that have"
|
||||||
|
echo " child datasets. Backups then read from a frozen, complete staging"
|
||||||
|
echo " tree instead of live files."
|
||||||
|
echo ""
|
||||||
|
echo " This changes how your backups read their source data. Verify that a"
|
||||||
|
echo " backup completes AND that its restic snapshot actually contains"
|
||||||
|
echo " child-dataset data before you rely on it."
|
||||||
|
;;
|
||||||
|
off)
|
||||||
|
if [ -f "$_NESTED_MARKER" ]; then
|
||||||
|
rm -f "$_NESTED_MARKER"
|
||||||
|
# Tear down any staging tree first: those bind mounts PIN their ZFS
|
||||||
|
# snapshots, so leaving them would block those snapshots from ever
|
||||||
|
# being destroyed. apply.sh (below) then reverts the patched files.
|
||||||
|
python3 "$PATCH_DIR/patch/truecloud_nested.py" cleanup || \
|
||||||
|
echo " WARNING: staging mounts remain; unmount them manually."
|
||||||
|
echo "Nested-dataset snapshots: DISABLED."
|
||||||
|
echo " apply.sh will revert the patched middleware files and the stock"
|
||||||
|
echo " guard is restored when middlewared restarts (this script does that)."
|
||||||
|
echo " Any task that already has snapshot=true on a nested dataset will"
|
||||||
|
echo " fail validation on its next edit. Turn the option off on those"
|
||||||
|
echo " tasks first, or re-run with --enable-nested-snapshots."
|
||||||
|
else
|
||||||
|
echo "Nested-dataset snapshots: already disabled."
|
||||||
|
fi
|
||||||
|
;;
|
||||||
|
*)
|
||||||
|
if [ -f "$_NESTED_MARKER" ]; then
|
||||||
|
echo "Nested-dataset snapshots: enabled (unchanged)."
|
||||||
|
else
|
||||||
|
echo "Nested-dataset snapshots: disabled (default)."
|
||||||
|
echo " Enable with: bash install.sh --enable-nested-snapshots"
|
||||||
|
fi
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
echo ""
|
||||||
|
|
||||||
# ── Apply now ─────────────────────────────────────────────────────────────────
|
# ── Apply now ─────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
echo "Applying patches ..."
|
echo "Applying patches ..."
|
||||||
|
|||||||
Binary file not shown.
+452
-63
@@ -7,13 +7,21 @@
|
|||||||
# is already up and has already imported the stock modules — the on-disk
|
# is already up and has already imported the stock modules — the on-disk
|
||||||
# patch alone cannot reach the running process.
|
# patch alone cannot reach the running process.
|
||||||
#
|
#
|
||||||
# TrueNAS updates replace /usr/ entirely; this script re-applies two patches:
|
# TrueNAS updates replace /usr/ entirely; this script re-applies three patches:
|
||||||
#
|
#
|
||||||
# 1. Backend — b2.py and restic.py are patched directly in the overlay.
|
# 1. Backend — b2.py and restic.py are patched directly in the overlay.
|
||||||
# On a boot run, a single detached middlewared restart is scheduled
|
# On a boot run, a single detached middlewared restart is scheduled
|
||||||
# (Step 3) so the patched modules actually get loaded.
|
# (Step 3) so the patched modules actually get loaded.
|
||||||
#
|
#
|
||||||
# 2. Angular JS bundle — Widens the TrueCloud Backup credential dropdown
|
# 2. Nested-dataset snapshots — installs _truecloud_nested.py and patches
|
||||||
|
# plugins/cloud/{snapshot,crud}.py + plugins/cloud_backup/sync.py so the
|
||||||
|
# "Take Snapshot" option works on a dataset that has child datasets.
|
||||||
|
# Stock middleware refuses that config, because it points the backup tool
|
||||||
|
# at the PARENT's .zfs/snapshot/ where children are invisible — it would
|
||||||
|
# silently back up a near-empty tree. We stage a complete tree of
|
||||||
|
# per-dataset bind mounts and only then relax the guard.
|
||||||
|
#
|
||||||
|
# 3. Angular JS bundle — Widens the TrueCloud Backup credential dropdown
|
||||||
# from Storj-only to include S3 and B2. Served from
|
# from Storj-only to include S3 and B2. Served from
|
||||||
# disk per request, so no restart is needed for it.
|
# disk per request, so no restart is needed for it.
|
||||||
#
|
#
|
||||||
@@ -24,7 +32,7 @@
|
|||||||
# Derive PATCH_DIR from this script's location (parent of the patch/ directory).
|
# Derive PATCH_DIR from this script's location (parent of the patch/ directory).
|
||||||
PATCH_DIR="$(cd "$(dirname "$0")/.." && pwd)"
|
PATCH_DIR="$(cd "$(dirname "$0")/.." && pwd)"
|
||||||
LOG="$PATCH_DIR/apply.log"
|
LOG="$PATCH_DIR/apply.log"
|
||||||
VERSION="0.0.4"
|
VERSION="0.3.2"
|
||||||
|
|
||||||
# Rotate log at 512 KB to avoid unbounded growth on a system volume.
|
# Rotate log at 512 KB to avoid unbounded growth on a system volume.
|
||||||
# Keep two prior generations (.1 and .2) so the last three boots are always available.
|
# Keep two prior generations (.1 and .2) so the last three boots are always available.
|
||||||
@@ -101,10 +109,20 @@ echo "Using Python: $PYTHON"
|
|||||||
# Combines what were previously four separate Python invocations into one to
|
# Combines what were previously four separate Python invocations into one to
|
||||||
# avoid repeated interpreter startup overhead under the PREINIT timeout budget.
|
# avoid repeated interpreter startup overhead under the PREINIT timeout budget.
|
||||||
|
|
||||||
|
# The patch has two independent modules, and each retires on its own:
|
||||||
|
#
|
||||||
|
# providers — B2/S3 credentials for TrueCloud Backup (b2.py, restic.py, UI)
|
||||||
|
# nested — snapshots on datasets that have child datasets (cloud/*.py)
|
||||||
|
#
|
||||||
|
# TrueNAS may well ship one natively long before the other, so a single
|
||||||
|
# all-or-nothing kill switch would silently take a still-needed module down with
|
||||||
|
# the superseded one. Each module is detected separately and skipped on its own;
|
||||||
|
# the global kill switch fires only once BOTH are native.
|
||||||
|
|
||||||
_tc_info=$("$PYTHON" -c "
|
_tc_info=$("$PYTHON" -c "
|
||||||
import inspect, os, sys
|
import inspect, os, sys
|
||||||
|
|
||||||
result = {'native': 'no', 'site_pkg': '', 'mw_dir': ''}
|
result = {'native_b2': 'no', 'native_nested': 'no', 'site_pkg': '', 'mw_dir': ''}
|
||||||
|
|
||||||
try:
|
try:
|
||||||
import middlewared
|
import middlewared
|
||||||
@@ -118,6 +136,7 @@ except ImportError:
|
|||||||
except Exception:
|
except Exception:
|
||||||
pass
|
pass
|
||||||
|
|
||||||
|
# providers: does B2RcloneRemote already carry a real get_restic_config()?
|
||||||
try:
|
try:
|
||||||
import middlewared.rclone.remote.b2 as _b2_mod
|
import middlewared.rclone.remote.b2 as _b2_mod
|
||||||
from middlewared.rclone.remote.b2 import B2RcloneRemote
|
from middlewared.rclone.remote.b2 import B2RcloneRemote
|
||||||
@@ -129,21 +148,79 @@ try:
|
|||||||
except (OSError, TypeError):
|
except (OSError, TypeError):
|
||||||
method_src = ''
|
method_src = ''
|
||||||
if 'NotImplementedError' not in method_src:
|
if 'NotImplementedError' not in method_src:
|
||||||
result['native'] = 'yes'
|
result['native_b2'] = 'yes'
|
||||||
except Exception:
|
except Exception:
|
||||||
pass
|
pass
|
||||||
|
|
||||||
print(result['native'])
|
# nested: stock gates nested datasets with a validation in plugins/cloud/crud.py.
|
||||||
|
# If that guard is gone, iX removed it, which means they implemented the traversal.
|
||||||
|
#
|
||||||
|
# Only look at the STOCK part of the file. Our own CRUD_BLOCK quotes the guard
|
||||||
|
# message (it filters on it), so scanning the whole file would find the string in
|
||||||
|
# our own patch and conclude the guard is still there. That happens to be
|
||||||
|
# harmless today because detection runs before patching, but it makes the probe
|
||||||
|
# silently order-dependent -- so cut our block off explicitly.
|
||||||
|
#
|
||||||
|
# If the file cannot be read we assume 'no' and keep patching: worst case the
|
||||||
|
# patch declines to apply and the option simply stays unavailable.
|
||||||
|
try:
|
||||||
|
crud = os.path.join(result['mw_dir'], 'plugins', 'cloud', 'crud.py')
|
||||||
|
with open(crud, encoding='utf-8', errors='replace') as fh:
|
||||||
|
stock_src = fh.read().split('\n# TRUECLOUD_PATCH', 1)[0]
|
||||||
|
# The guard message is SPLIT across adjacent string literals in the source:
|
||||||
|
#
|
||||||
|
# verrors.add(..., 'This option is only available for datasets that have no further '
|
||||||
|
# 'nesting')
|
||||||
|
#
|
||||||
|
# Python concatenates those at runtime, so the errmsg is contiguous -- but the
|
||||||
|
# SOURCE never contains the whole phrase. A raw search finds nothing, concludes
|
||||||
|
# iX removed the guard, and silently skips this module FOREVER. (Caught only by
|
||||||
|
# running the probe against real middlewared.)
|
||||||
|
#
|
||||||
|
# Strip whitespace and quote characters, then match the compacted phrase. That
|
||||||
|
# is robust to any wrapping or concatenation style iX may use.
|
||||||
|
_drop = str.maketrans('', '', ' \\t\\n\\r' + chr(34) + chr(39))
|
||||||
|
if 'nofurthernesting' not in stock_src.translate(_drop):
|
||||||
|
result['native_nested'] = 'yes'
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
print(result['native_b2'])
|
||||||
|
print(result['native_nested'])
|
||||||
print(result['site_pkg'])
|
print(result['site_pkg'])
|
||||||
print(result['mw_dir'])
|
print(result['mw_dir'])
|
||||||
" 2>/dev/null || printf 'no\n\n\n')
|
" 2>/dev/null || printf 'no\nno\n\n\n')
|
||||||
|
|
||||||
_tc_native=$(printf '%s' "$_tc_info" | sed -n '1p')
|
_tc_native_b2=$(printf '%s' "$_tc_info" | sed -n '1p')
|
||||||
SITE_PKG=$(printf '%s' "$_tc_info" | sed -n '2p')
|
_tc_native_nested=$(printf '%s' "$_tc_info" | sed -n '2p')
|
||||||
_MW_DIR=$(printf '%s' "$_tc_info" | sed -n '3p')
|
SITE_PKG=$(printf '%s' "$_tc_info" | sed -n '3p')
|
||||||
|
_MW_DIR=$(printf '%s' "$_tc_info" | sed -n '4p')
|
||||||
|
|
||||||
if [ "$_tc_native" = "yes" ]; then
|
# Nested support is opt-in; if it was never enabled, it cannot be the reason to
|
||||||
echo "NOTICE: TrueNAS now provides native B2 restic support — truecloud-patch is no longer needed."
|
# keep the patch alive.
|
||||||
|
if [ -f "$PATCH_DIR/nested_snapshots_enabled" ]; then
|
||||||
|
_NESTED_ENABLED=1
|
||||||
|
else
|
||||||
|
_NESTED_ENABLED=0
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Is either module still doing something useful?
|
||||||
|
_providers_needed=1
|
||||||
|
[ "$_tc_native_b2" = "yes" ] && _providers_needed=0
|
||||||
|
|
||||||
|
_nested_needed=0
|
||||||
|
if [ "$_NESTED_ENABLED" = "1" ] && [ "$_tc_native_nested" != "yes" ]; then
|
||||||
|
_nested_needed=1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [ "$_providers_needed" = "0" ] && [ "$_nested_needed" = "0" ]; then
|
||||||
|
echo "NOTICE: Nothing left for truecloud-patch to do:"
|
||||||
|
[ "$_tc_native_b2" = "yes" ] && echo "NOTICE: - TrueNAS now provides native B2 restic support."
|
||||||
|
if [ "$_tc_native_nested" = "yes" ]; then
|
||||||
|
echo "NOTICE: - TrueNAS now handles snapshots on nested datasets natively."
|
||||||
|
elif [ "$_NESTED_ENABLED" = "0" ]; then
|
||||||
|
echo "NOTICE: - Nested-dataset snapshots are not enabled (opt-in)."
|
||||||
|
fi
|
||||||
echo "NOTICE: Setting kill switch; patching will be skipped on all future boots."
|
echo "NOTICE: Setting kill switch; patching will be skipped on all future boots."
|
||||||
echo "NOTICE: Run the following to fully remove the patch:"
|
echo "NOTICE: Run the following to fully remove the patch:"
|
||||||
echo "NOTICE: bash $PATCH_DIR/uninstall.sh"
|
echo "NOTICE: bash $PATCH_DIR/uninstall.sh"
|
||||||
@@ -158,12 +235,21 @@ if [ "$_tc_native" = "yes" ]; then
|
|||||||
exit 0
|
exit 0
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
if [ "$_providers_needed" = "0" ]; then
|
||||||
|
echo "NOTICE: TrueNAS now provides native B2 restic support — the providers module"
|
||||||
|
echo "NOTICE: is superseded and will be skipped. The nested-snapshot module is still"
|
||||||
|
echo "NOTICE: active, so the patch stays installed."
|
||||||
|
fi
|
||||||
|
if [ "$_NESTED_ENABLED" = "1" ] && [ "$_tc_native_nested" = "yes" ]; then
|
||||||
|
echo "NOTICE: TrueNAS now handles nested-dataset snapshots natively — that module is"
|
||||||
|
echo "NOTICE: superseded and will be skipped. You can drop --enable-nested-snapshots."
|
||||||
|
fi
|
||||||
|
|
||||||
# ── Step 1: backend patch ─────────────────────────────────────────────────────
|
# ── Step 1: backend patch ─────────────────────────────────────────────────────
|
||||||
|
|
||||||
echo "--- backend patch ---"
|
echo "--- backend patch ---"
|
||||||
|
|
||||||
_b2_ok=0
|
_backend_ok=0
|
||||||
_restic_ok=0
|
|
||||||
|
|
||||||
if [ -z "$SITE_PKG" ]; then
|
if [ -z "$SITE_PKG" ]; then
|
||||||
echo "WARNING: Cannot determine site-packages directory; skipping backend patch."
|
echo "WARNING: Cannot determine site-packages directory; skipping backend patch."
|
||||||
@@ -175,12 +261,27 @@ elif [ -z "$_MW_DIR" ]; then
|
|||||||
else
|
else
|
||||||
_B2_PY="$_MW_DIR/rclone/remote/b2.py"
|
_B2_PY="$_MW_DIR/rclone/remote/b2.py"
|
||||||
_RESTIC_PY="$_MW_DIR/plugins/cloud_backup/restic.py"
|
_RESTIC_PY="$_MW_DIR/plugins/cloud_backup/restic.py"
|
||||||
|
_CLOUD_DIR="$_MW_DIR/plugins/cloud"
|
||||||
|
_SYNC_PY="$_MW_DIR/plugins/cloud_backup/sync.py"
|
||||||
|
_NESTED_SRC="$PATCH_DIR/patch/truecloud_nested.py"
|
||||||
|
|
||||||
# ── patch b2.py + restic.py + hook_status.json (single subprocess) ──────
|
# Each module is applied only if it is still needed. _providers_needed and
|
||||||
if "$PYTHON" - "$_B2_PY" "$_RESTIC_PY" "$PATCH_DIR/hook_status.json" << 'PYEOF'
|
# _nested_needed were computed above (native-support detection + the opt-in
|
||||||
import json, os, sys, time
|
# marker), so one module going native never disables the other.
|
||||||
|
# ── patch b2.py + restic.py + nested-snapshot + hook_status.json ────────
|
||||||
|
# (single subprocess: PREINIT has a tight timeout budget)
|
||||||
|
if "$PYTHON" - "$_B2_PY" "$_RESTIC_PY" "$PATCH_DIR/hook_status.json" \
|
||||||
|
"$_CLOUD_DIR" "$_SYNC_PY" "$_NESTED_SRC" \
|
||||||
|
"$_providers_needed" "$_nested_needed" "$_NESTED_ENABLED" \
|
||||||
|
"$_tc_native_nested" << 'PYEOF'
|
||||||
|
import json, os, shutil, sys, time
|
||||||
|
|
||||||
b2_path, restic_path, status_path = sys.argv[1], sys.argv[2], sys.argv[3]
|
b2_path, restic_path, status_path = sys.argv[1], sys.argv[2], sys.argv[3]
|
||||||
|
cloud_dir, sync_path, nested_src = sys.argv[4], sys.argv[5], sys.argv[6]
|
||||||
|
providers_needed = sys.argv[7] == "1"
|
||||||
|
nested_needed = sys.argv[8] == "1"
|
||||||
|
nested_enabled = sys.argv[9] == "1"
|
||||||
|
nested_native = sys.argv[10] == "yes"
|
||||||
|
|
||||||
B2_BLOCK = """
|
B2_BLOCK = """
|
||||||
# TRUECLOUD_PATCH — added by truenas-truecloud-patch/patch/apply.sh
|
# TRUECLOUD_PATCH — added by truenas-truecloud-patch/patch/apply.sh
|
||||||
@@ -240,6 +341,133 @@ else:
|
|||||||
get_restic_config._truecloud_patched = True
|
get_restic_config._truecloud_patched = True
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
# ── nested-dataset snapshot support ───────────────────────────────────────────
|
||||||
|
# Stock middleware refuses `snapshot=true` on a path containing child datasets,
|
||||||
|
# because it points the backup tool at the PARENT dataset's .zfs/snapshot/,
|
||||||
|
# where child datasets are INVISIBLE -- it would silently back up a near-empty
|
||||||
|
# tree. That guard is correct. We implement the missing traversal (a staging
|
||||||
|
# tree of per-dataset bind mounts) and only then relax the guard.
|
||||||
|
#
|
||||||
|
# Fail-safe direction: if any of these three blocks fails to apply, the stock
|
||||||
|
# guard remains and the option simply stays unavailable. We never end up with
|
||||||
|
# the guard removed but the traversal missing -- that would be a silently empty
|
||||||
|
# backup, the worst possible outcome.
|
||||||
|
|
||||||
|
SNAPSHOT_BLOCK = """
|
||||||
|
# TRUECLOUD_PATCH — added by truenas-truecloud-patch/patch/apply.sh
|
||||||
|
try:
|
||||||
|
from middlewared.plugins.cloud import _truecloud_nested as _tc_nested
|
||||||
|
except ImportError:
|
||||||
|
_tc_nested = None
|
||||||
|
|
||||||
|
if _tc_nested is not None:
|
||||||
|
_tc_orig_create_snapshot = create_snapshot
|
||||||
|
|
||||||
|
async def create_snapshot(middleware, path, name="cloud_task-onetime"):
|
||||||
|
# Stock takes the (already recursive) snapshot; we only replace the PATH.
|
||||||
|
snapshot, snap_path = await _tc_orig_create_snapshot(middleware, path, name)
|
||||||
|
|
||||||
|
_logger = getattr(middleware, "logger", None)
|
||||||
|
try:
|
||||||
|
# Enumerate datasets AFTER the snapshot, never before. The snapshot is
|
||||||
|
# the point-in-time truth; a list read beforehand could miss a dataset
|
||||||
|
# created in the gap, which the recursive snapshot WOULD capture but
|
||||||
|
# our staging plan would not -- silently omitting it from the backup.
|
||||||
|
# Read afterwards, an unsnapshotted dataset instead trips the isdir()
|
||||||
|
# check in plan_staging and fails the run loudly. Loud beats silent.
|
||||||
|
datasets = await middleware.call(
|
||||||
|
"zfs.dataset.query", [["type", "=", "FILESYSTEM"]]
|
||||||
|
)
|
||||||
|
dataset, nested = get_dataset_recursive(datasets, path)
|
||||||
|
|
||||||
|
if not nested:
|
||||||
|
# No children: stock behaviour, untouched. Stock's `finally` owns
|
||||||
|
# the snapshot from here (its non-recursive delete is correct,
|
||||||
|
# because a non-nested snapshot has no children).
|
||||||
|
return snapshot, snap_path
|
||||||
|
|
||||||
|
staging_root = await _tc_nested.stage_nested(
|
||||||
|
middleware, path, snapshot,
|
||||||
|
dataset["name"], dataset["properties"]["mountpoint"]["value"],
|
||||||
|
name, datasets, logger=_logger,
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
# The snapshot exists, but this exception means sync.py never completes
|
||||||
|
# `snapshot, local_path = await create_snapshot(...)`, so its local
|
||||||
|
# `snapshot` stays None and its `finally` deletes NOTHING. Sweep the
|
||||||
|
# tree ourselves or leak the parent plus one snapshot per descendant
|
||||||
|
# dataset (160+ here) on every failed run.
|
||||||
|
await _tc_nested.delete_snapshot_tree(middleware, snapshot, logger=_logger)
|
||||||
|
raise
|
||||||
|
|
||||||
|
return snapshot, staging_root
|
||||||
|
|
||||||
|
create_snapshot._truecloud_patched = True
|
||||||
|
"""
|
||||||
|
|
||||||
|
CRUD_BLOCK = """
|
||||||
|
# TRUECLOUD_PATCH — added by truenas-truecloud-patch/patch/apply.sh
|
||||||
|
try:
|
||||||
|
from middlewared.plugins.cloud import _truecloud_nested as _tc_nested
|
||||||
|
except ImportError:
|
||||||
|
_tc_nested = None
|
||||||
|
|
||||||
|
if _tc_nested is not None:
|
||||||
|
_tc_orig_validate = CloudTaskServiceMixin._validate
|
||||||
|
|
||||||
|
async def _tc_validate(self, app, verrors, name, data):
|
||||||
|
await _tc_orig_validate(self, app, verrors, name, data)
|
||||||
|
|
||||||
|
# Only cloud_backup: staging teardown is wired into cloud_backup.sync's
|
||||||
|
# finally. cloudsync would leak bind mounts, so leave its guard intact.
|
||||||
|
if getattr(getattr(self, "_config", None), "namespace", "") != "cloud_backup":
|
||||||
|
return
|
||||||
|
|
||||||
|
# Drop ONLY the nested-dataset guard. If iX ever rewords the message the
|
||||||
|
# filter stops matching, the guard survives, and the option merely stays
|
||||||
|
# unavailable -- the safe direction to fail.
|
||||||
|
verrors.errors = [
|
||||||
|
e for e in verrors.errors
|
||||||
|
if not (
|
||||||
|
getattr(e, "attribute", "") == f"{name}.snapshot"
|
||||||
|
and "no further nesting" in getattr(e, "errmsg", "")
|
||||||
|
)
|
||||||
|
]
|
||||||
|
|
||||||
|
CloudTaskServiceMixin._validate = _tc_validate
|
||||||
|
CloudTaskServiceMixin._validate._truecloud_patched = True
|
||||||
|
"""
|
||||||
|
|
||||||
|
SYNC_BLOCK = """
|
||||||
|
# TRUECLOUD_PATCH — added by truenas-truecloud-patch/patch/apply.sh
|
||||||
|
try:
|
||||||
|
from middlewared.plugins.cloud import _truecloud_nested as _tc_nested
|
||||||
|
except ImportError:
|
||||||
|
_tc_nested = None
|
||||||
|
|
||||||
|
if _tc_nested is not None:
|
||||||
|
_tc_orig_restic_backup = restic_backup
|
||||||
|
|
||||||
|
async def restic_backup(middleware, job, cloud_backup, dry_run=False, rate_limit=None):
|
||||||
|
# Our bind mounts pin the ZFS snapshot, so stock's `finally` cannot
|
||||||
|
# destroy it (EBUSY) and logs one benign warning. We unmount here and
|
||||||
|
# then delete the snapshot for real.
|
||||||
|
try:
|
||||||
|
return await _tc_orig_restic_backup(middleware, job, cloud_backup, dry_run, rate_limit)
|
||||||
|
finally:
|
||||||
|
try:
|
||||||
|
await _tc_nested.cleanup_task(
|
||||||
|
middleware,
|
||||||
|
f"cloud_backup-{cloud_backup.get('id', 'onetime')}",
|
||||||
|
logger=getattr(middleware, "logger", None),
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
middleware.logger.warning("truecloud-patch: staging cleanup failed: %r", e)
|
||||||
|
|
||||||
|
restic_backup._truecloud_patched = True
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
def patch_file(path, block):
|
def patch_file(path, block):
|
||||||
with open(path, encoding="utf-8") as fh:
|
with open(path, encoding="utf-8") as fh:
|
||||||
content = fh.read()
|
content = fh.read()
|
||||||
@@ -249,36 +477,158 @@ def patch_file(path, block):
|
|||||||
with open(path, "w", encoding="utf-8") as fh:
|
with open(path, "w", encoding="utf-8") as fh:
|
||||||
fh.write(base.rstrip("\n") + "\n" + block)
|
fh.write(base.rstrip("\n") + "\n" + block)
|
||||||
|
|
||||||
|
|
||||||
|
def unpatch_file(path):
|
||||||
|
"""Strip our appended block, restoring the stock file. True if it was patched."""
|
||||||
|
try:
|
||||||
|
with open(path, encoding="utf-8") as fh:
|
||||||
|
content = fh.read()
|
||||||
|
except OSError:
|
||||||
|
return False
|
||||||
|
idx = content.find("\n# TRUECLOUD_PATCH")
|
||||||
|
if idx == -1:
|
||||||
|
return False
|
||||||
|
try:
|
||||||
|
with open(path, "w", encoding="utf-8") as fh:
|
||||||
|
fh.write(content[:idx].rstrip("\n") + "\n")
|
||||||
|
except OSError:
|
||||||
|
return False
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def revert_nested(cloud_dir, sync_path):
|
||||||
|
"""Undo the nested patch. Returns the names of what was actually reverted.
|
||||||
|
|
||||||
|
Skipping the patch is NOT enough to disable the feature. The overlay persists
|
||||||
|
for the whole boot, so an earlier run this boot may already have written the
|
||||||
|
patched files -- and middlewared re-imports them on the restart that
|
||||||
|
install.sh performs. Without this, `install.sh --disable-nested-snapshots`
|
||||||
|
would report "disabled" while the feature kept running until the next reboot.
|
||||||
|
"""
|
||||||
|
reverted = []
|
||||||
|
|
||||||
|
# Remove the module FIRST. Every injected block is guarded by
|
||||||
|
# `if _tc_nested is not None`, so once it is gone they all no-op even if a
|
||||||
|
# later step here fails -- the guard is restored no matter what.
|
||||||
|
try:
|
||||||
|
os.unlink(os.path.join(cloud_dir, '_truecloud_nested.py'))
|
||||||
|
reverted.append('_truecloud_nested.py')
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
# NB: restic.py also carries a TRUECLOUD_PATCH block, but that belongs to the
|
||||||
|
# providers module. Only these three are ours to revert.
|
||||||
|
for name, path in (
|
||||||
|
('crud.py', os.path.join(cloud_dir, 'crud.py')),
|
||||||
|
('sync.py', sync_path),
|
||||||
|
('snapshot.py', os.path.join(cloud_dir, 'snapshot.py')),
|
||||||
|
):
|
||||||
|
if unpatch_file(path):
|
||||||
|
reverted.append(name)
|
||||||
|
|
||||||
|
return reverted
|
||||||
|
|
||||||
b2_ok = restic_ok = False
|
b2_ok = restic_ok = False
|
||||||
|
nested_ok = False
|
||||||
|
|
||||||
if os.path.exists(b2_path):
|
# ── module: providers (B2/S3) ─────────────────────────────────────────────────
|
||||||
try:
|
# Skipped entirely once TrueNAS ships native B2 restic support. That must not
|
||||||
patch_file(b2_path, B2_BLOCK)
|
# take the nested module down with it, so the two are gated independently.
|
||||||
b2_ok = True
|
providers_detail = ''
|
||||||
print(f"OK: Patched b2.py → {b2_path}")
|
if not providers_needed:
|
||||||
except Exception as e:
|
providers_detail = 'superseded: TrueNAS provides native B2 restic support'
|
||||||
print(f"WARNING: Failed to patch b2.py: {e}")
|
print('INFO: Providers module skipped — TrueNAS now supports B2 natively.')
|
||||||
else:
|
else:
|
||||||
print(f"WARNING: b2.py not found at {b2_path}")
|
if os.path.exists(b2_path):
|
||||||
|
try:
|
||||||
|
patch_file(b2_path, B2_BLOCK)
|
||||||
|
b2_ok = True
|
||||||
|
print(f"OK: Patched b2.py → {b2_path}")
|
||||||
|
except Exception as e:
|
||||||
|
print(f"WARNING: Failed to patch b2.py: {e}")
|
||||||
|
else:
|
||||||
|
print(f"WARNING: b2.py not found at {b2_path}")
|
||||||
|
|
||||||
if os.path.exists(restic_path):
|
if os.path.exists(restic_path):
|
||||||
try:
|
try:
|
||||||
patch_file(restic_path, RESTIC_BLOCK)
|
patch_file(restic_path, RESTIC_BLOCK)
|
||||||
restic_ok = True
|
restic_ok = True
|
||||||
print(f"OK: Patched restic.py → {restic_path}")
|
print(f"OK: Patched restic.py → {restic_path}")
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
print(f"WARNING: Failed to patch restic.py: {e}")
|
print(f"WARNING: Failed to patch restic.py: {e}")
|
||||||
|
else:
|
||||||
|
print(f"WARNING: restic.py not found at {restic_path}")
|
||||||
|
providers_detail = (
|
||||||
|
'patched on disk in overlay at boot' if (b2_ok and restic_ok)
|
||||||
|
else 'b2.py/restic.py not found or write failed'
|
||||||
|
)
|
||||||
|
|
||||||
|
# ── module: nested-dataset snapshots ──────────────────────────────────────────
|
||||||
|
# Order matters: install the traversal machinery FIRST, relax the validation
|
||||||
|
# guard LAST. If anything fails partway, the guard is still in place and the
|
||||||
|
# option stays unavailable -- we never expose "guard removed, traversal missing".
|
||||||
|
nested_detail = ''
|
||||||
|
if not nested_needed:
|
||||||
|
# Not just "skip": actively revert. The overlay lives for the whole boot, so a
|
||||||
|
# previously-applied patch is still sitting there and middlewared would
|
||||||
|
# re-import it on restart. See revert_nested().
|
||||||
|
if not nested_enabled:
|
||||||
|
nested_detail = 'disabled (opt-in; enable with: install.sh --enable-nested-snapshots)'
|
||||||
|
print('INFO: Nested-dataset snapshot support is disabled (opt-in feature).')
|
||||||
|
elif nested_native:
|
||||||
|
nested_detail = 'superseded: TrueNAS handles nested-dataset snapshots natively'
|
||||||
|
print('INFO: Nested module skipped — TrueNAS now handles nesting natively.')
|
||||||
|
else:
|
||||||
|
nested_detail = 'not needed'
|
||||||
|
print('INFO: Nested module skipped.')
|
||||||
|
|
||||||
|
reverted = revert_nested(cloud_dir, sync_path)
|
||||||
|
if reverted:
|
||||||
|
print('OK: Reverted a previously-applied nested patch (' + ', '.join(reverted) + ').')
|
||||||
|
print(' The stock nesting guard is restored once middlewared restarts.')
|
||||||
|
nested_detail += ' — previous patch reverted'
|
||||||
|
|
||||||
|
if not nested_enabled:
|
||||||
|
print('INFO: Enable with: bash install.sh --enable-nested-snapshots')
|
||||||
else:
|
else:
|
||||||
print(f"WARNING: restic.py not found at {restic_path}")
|
try:
|
||||||
|
snapshot_py = os.path.join(cloud_dir, 'snapshot.py')
|
||||||
|
crud_py = os.path.join(cloud_dir, 'crud.py')
|
||||||
|
nested_dst = os.path.join(cloud_dir, '_truecloud_nested.py')
|
||||||
|
|
||||||
|
missing = [p for p in (snapshot_py, crud_py, sync_path, nested_src) if not os.path.exists(p)]
|
||||||
|
if missing:
|
||||||
|
raise FileNotFoundError('missing: ' + ', '.join(missing))
|
||||||
|
|
||||||
|
shutil.copyfile(nested_src, nested_dst) # 1. traversal implementation
|
||||||
|
patch_file(snapshot_py, SNAPSHOT_BLOCK) # 2. build the staging tree
|
||||||
|
patch_file(sync_path, SYNC_BLOCK) # 3. tear it down afterwards
|
||||||
|
patch_file(crud_py, CRUD_BLOCK) # 4. ONLY NOW allow nested tasks
|
||||||
|
|
||||||
|
nested_ok = True
|
||||||
|
nested_detail = 'nested-dataset snapshots enabled (staging tree)'
|
||||||
|
print(f'OK: Installed nested-snapshot support → {nested_dst}')
|
||||||
|
print(f'OK: Patched snapshot.py, sync.py, crud.py → {cloud_dir}')
|
||||||
|
except Exception as e:
|
||||||
|
nested_detail = f'not applied: {e}'
|
||||||
|
print(f'WARNING: Failed to apply nested-snapshot patch: {e}')
|
||||||
|
print('WARNING: Stock nesting guard remains; snapshot option stays unavailable')
|
||||||
|
print('WARNING: for nested datasets. Existing backups are unaffected.')
|
||||||
|
|
||||||
|
# One entry per MODULE, not per file. `ok` means "nothing is wrong", so a module
|
||||||
|
# that is inactive (superseded, or opt-in and off) is ok -- reporting a disabled
|
||||||
|
# opt-in feature as FAIL would make `create_task.py verify` fail on a default
|
||||||
|
# install. `active` says whether the module is doing anything.
|
||||||
patches = {
|
patches = {
|
||||||
'middlewared.rclone.remote.b2': {
|
'providers': {
|
||||||
'ok': b2_ok,
|
'ok': (not providers_needed) or bool(b2_ok and restic_ok),
|
||||||
'detail': 'patched on disk in overlay at boot' if b2_ok else 'b2.py not found or write failed',
|
'active': providers_needed,
|
||||||
|
'detail': providers_detail,
|
||||||
},
|
},
|
||||||
'middlewared.plugins.cloud_backup.restic': {
|
'nested_snapshots': {
|
||||||
'ok': restic_ok,
|
'ok': (not nested_needed) or nested_ok,
|
||||||
'detail': 'patched on disk in overlay at boot' if restic_ok else 'restic.py not found or write failed',
|
'active': nested_needed,
|
||||||
|
'detail': nested_detail,
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
payload = {'patched_at': time.strftime('%Y-%m-%dT%H:%M:%SZ', time.gmtime()), 'patches': patches}
|
payload = {'patched_at': time.strftime('%Y-%m-%dT%H:%M:%SZ', time.gmtime()), 'patches': patches}
|
||||||
@@ -291,35 +641,61 @@ try:
|
|||||||
except OSError as e:
|
except OSError as e:
|
||||||
print(f'WARNING: Could not write hook_status.json: {e}')
|
print(f'WARNING: Could not write hook_status.json: {e}')
|
||||||
|
|
||||||
sys.exit(0 if (b2_ok and restic_ok) else 1)
|
# Exit code tells the caller whether a middlewared restart is still worth doing:
|
||||||
|
#
|
||||||
|
# 0 every module that was needed applied cleanly
|
||||||
|
# 2 PARTIAL -- one module failed but another landed, so there IS something new
|
||||||
|
# on disk waiting to be loaded
|
||||||
|
# 1 nothing landed; a restart would accomplish nothing
|
||||||
|
#
|
||||||
|
# Collapsing 2 into 1 would mean a failing providers patch suppresses the restart
|
||||||
|
# that a freshly-applied nested patch needs, leaving it on disk and never loaded.
|
||||||
|
_providers_done = (not providers_needed) or bool(b2_ok and restic_ok)
|
||||||
|
_nested_done = (not nested_needed) or nested_ok
|
||||||
|
_landed = (providers_needed and b2_ok and restic_ok) or (nested_needed and nested_ok)
|
||||||
|
|
||||||
|
if _providers_done and _nested_done:
|
||||||
|
sys.exit(0)
|
||||||
|
sys.exit(2 if _landed else 1)
|
||||||
PYEOF
|
PYEOF
|
||||||
then
|
then
|
||||||
_b2_ok=1
|
_backend_ok=1
|
||||||
_restic_ok=1
|
|
||||||
else
|
else
|
||||||
# Individual results already printed above; exit code 1 means at least one failed.
|
_rc=$?
|
||||||
true
|
if [ "$_rc" = "2" ]; then
|
||||||
|
# One module failed, but another was applied and still needs loading.
|
||||||
|
_backend_ok=1
|
||||||
|
echo "WARNING: a module failed to apply; the other landed and will be loaded."
|
||||||
|
else
|
||||||
|
_backend_ok=0
|
||||||
|
fi
|
||||||
fi
|
fi
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# ── Step 2: Angular bundle ────────────────────────────────────────────────────
|
# ── Step 2: Angular bundle ────────────────────────────────────────────────────
|
||||||
|
# Belongs to the providers module (it widens the credential dropdown), so it is
|
||||||
|
# skipped along with it once TrueNAS supports B2 natively.
|
||||||
|
|
||||||
echo "--- UI patch ---"
|
echo "--- UI patch ---"
|
||||||
|
|
||||||
# Ensure the webui directory is writable before patch_ui.py tries to create a
|
if [ "$_providers_needed" = "0" ]; then
|
||||||
# backup and write the patched bundle. On immutable OS we mount an overlay.
|
echo "Skipped — providers module superseded by native B2 support."
|
||||||
_webui_dir=""
|
else
|
||||||
for _d in /usr/share/truenas/webui /usr/share/truenas-ui /var/www/truenas; do
|
# Ensure the webui directory is writable before patch_ui.py tries to create a
|
||||||
if [ -d "$_d" ]; then
|
# backup and write the patched bundle. On immutable OS we mount an overlay.
|
||||||
_webui_dir="$_d"
|
_webui_dir=""
|
||||||
break
|
for _d in /usr/share/truenas/webui /usr/share/truenas-ui /var/www/truenas; do
|
||||||
|
if [ -d "$_d" ]; then
|
||||||
|
_webui_dir="$_d"
|
||||||
|
break
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
if [ -n "$_webui_dir" ]; then
|
||||||
|
_ensure_writable "$_webui_dir" "ui" || true # non-fatal; patch_ui.py reports the error
|
||||||
fi
|
fi
|
||||||
done
|
|
||||||
if [ -n "$_webui_dir" ]; then
|
|
||||||
_ensure_writable "$_webui_dir" "ui" || true # non-fatal; patch_ui.py reports the error
|
|
||||||
fi
|
|
||||||
|
|
||||||
"$PYTHON" "$PATCH_DIR/patch/patch_ui.py" || echo "WARNING: patch_ui.py exited non-zero; UI dropdown may still show Storj only."
|
"$PYTHON" "$PATCH_DIR/patch/patch_ui.py" || echo "WARNING: patch_ui.py exited non-zero; UI dropdown may still show Storj only."
|
||||||
|
fi
|
||||||
|
|
||||||
# ── Step 3: deferred middlewared restart (boot runs only) ─────────────────────
|
# ── Step 3: deferred middlewared restart (boot runs only) ─────────────────────
|
||||||
# At boot this script is spawned by middlewared, which already imported the
|
# At boot this script is spawned by middlewared, which already imported the
|
||||||
@@ -329,23 +705,36 @@ fi
|
|||||||
# ix-* boot units still need midclt to answer.
|
# ix-* boot units still need midclt to answer.
|
||||||
# Boot context is detected by the parent process being middlewared; manual
|
# Boot context is detected by the parent process being middlewared; manual
|
||||||
# runs (install.sh, recovery) never trigger a restart.
|
# runs (install.sh, recovery) never trigger a restart.
|
||||||
|
#
|
||||||
|
# The unit runs wait_restart.sh, which blocks until boot has actually
|
||||||
|
# settled (systemd job queue drained, docker/apps state terminal) before
|
||||||
|
# restarting. systemd ordering alone (After=multi-user.target, ≤ v0.0.4)
|
||||||
|
# fired while ix-reporting and the docker/apps startup were still in flight
|
||||||
|
# and killed both — apps and dashboard stats stayed down until the next
|
||||||
|
# boot. No Type=oneshot: a oneshot's start job would hold the boot queue
|
||||||
|
# open against the `is-system-running --wait` inside the script.
|
||||||
|
|
||||||
echo "--- deferred restart ---"
|
echo "--- deferred restart ---"
|
||||||
|
|
||||||
|
# Restart when ANY still-needed backend module landed (_backend_ok, incl. the
|
||||||
|
# partial case). Keying this off the providers module alone would skip the restart
|
||||||
|
# on a box where B2 has gone native but the nested module was freshly patched —
|
||||||
|
# leaving it on disk and never loaded.
|
||||||
|
#
|
||||||
|
# "No module active at all" cannot reach here: that is the kill-switch branch
|
||||||
|
# above, which exits.
|
||||||
if ! grep -aq middlewared "/proc/$PPID/cmdline" 2>/dev/null; then
|
if ! grep -aq middlewared "/proc/$PPID/cmdline" 2>/dev/null; then
|
||||||
echo "Manual run (parent is not middlewared) — no restart scheduled."
|
echo "Manual run (parent is not middlewared) — no restart scheduled."
|
||||||
elif [ "$_b2_ok" != "1" ] || [ "$_restic_ok" != "1" ]; then
|
elif [ "$_backend_ok" != "1" ]; then
|
||||||
echo "Backend patch incomplete — no restart scheduled (nothing new to load)."
|
echo "Nothing landed on disk — no restart scheduled (nothing new to load)."
|
||||||
else
|
else
|
||||||
# A failed unit from an earlier attempt this boot would block systemd-run.
|
# A failed unit from an earlier attempt this boot would block systemd-run.
|
||||||
systemctl reset-failed truecloud-mw-restart.service 2>/dev/null
|
systemctl reset-failed truecloud-mw-restart.service 2>/dev/null
|
||||||
if systemd-run --no-block --collect --unit=truecloud-mw-restart \
|
if systemd-run --no-block --collect --unit=truecloud-mw-restart \
|
||||||
--property=Type=oneshot \
|
/bin/bash "$PATCH_DIR/patch/wait_restart.sh"; then
|
||||||
--property=After=multi-user.target \
|
|
||||||
--property=After=ix-postinit.service \
|
|
||||||
systemctl try-restart middlewared; then
|
|
||||||
echo "OK: Scheduled deferred middlewared restart (unit: truecloud-mw-restart)."
|
echo "OK: Scheduled deferred middlewared restart (unit: truecloud-mw-restart)."
|
||||||
echo " Backend patch becomes active once boot completes."
|
echo " It waits for boot to fully settle (apps started, reporting up),"
|
||||||
|
echo " then restarts middlewared so the backend patch actually loads."
|
||||||
else
|
else
|
||||||
echo "WARNING: Could not schedule deferred restart — backend patch is on disk but NOT loaded."
|
echo "WARNING: Could not schedule deferred restart — backend patch is on disk but NOT loaded."
|
||||||
echo " Activate manually: systemctl restart middlewared"
|
echo " Activate manually: systemctl restart middlewared"
|
||||||
|
|||||||
+13
-1
@@ -127,13 +127,21 @@ def cmd_verify():
|
|||||||
print(f"Hook status (recorded at {status.get('patched_at', 'unknown')})")
|
print(f"Hook status (recorded at {status.get('patched_at', 'unknown')})")
|
||||||
print()
|
print()
|
||||||
all_ok = True
|
all_ok = True
|
||||||
|
any_active = False
|
||||||
for module, info in status.get("patches", {}).items():
|
for module, info in status.get("patches", {}).items():
|
||||||
ok = info.get("ok", False)
|
ok = info.get("ok", False)
|
||||||
|
# A module can be inactive because TrueNAS now does it natively, or
|
||||||
|
# because it is opt-in and switched off. Neither is a failure.
|
||||||
|
active = info.get("active", True)
|
||||||
label = "OK " if ok else "FAIL"
|
label = "OK " if ok else "FAIL"
|
||||||
|
if ok and not active:
|
||||||
|
label = "SKIP"
|
||||||
detail = f" — {info['detail']}" if info.get("detail") else ""
|
detail = f" — {info['detail']}" if info.get("detail") else ""
|
||||||
print(f" [{label}] {module}{detail}")
|
print(f" [{label}] {module}{detail}")
|
||||||
if not ok:
|
if not ok:
|
||||||
all_ok = False
|
all_ok = False
|
||||||
|
if active:
|
||||||
|
any_active = True
|
||||||
|
|
||||||
# The disk status alone can false-positive: at boot the files are patched
|
# The disk status alone can false-positive: at boot the files are patched
|
||||||
# while middlewared is already running with the stock modules imported.
|
# while middlewared is already running with the stock modules imported.
|
||||||
@@ -146,7 +154,11 @@ def cmd_verify():
|
|||||||
mw_start = _middlewared_start_epoch()
|
mw_start = _middlewared_start_epoch()
|
||||||
|
|
||||||
proc_stale = False
|
proc_stale = False
|
||||||
if patched_epoch is None or mw_start is None:
|
if not any_active:
|
||||||
|
# Nothing is patched into middlewared, so whether it restarted since is
|
||||||
|
# irrelevant -- there is nothing for it to have loaded.
|
||||||
|
print(" [-- ] running middlewared process — no active module; nothing to load")
|
||||||
|
elif patched_epoch is None or mw_start is None:
|
||||||
print(" [?? ] running middlewared process — could not compare start time;")
|
print(" [?? ] running middlewared process — could not compare start time;")
|
||||||
print(" the results above reflect the on-disk state only")
|
print(" the results above reflect the on-disk state only")
|
||||||
elif mw_start + 2 < patched_epoch:
|
elif mw_start + 2 < patched_epoch:
|
||||||
|
|||||||
+42
-10
@@ -9,8 +9,8 @@ in the minified JS in one of two forms depending on TrueNAS / Angular version:
|
|||||||
TrueNAS 24.x (static inline array):
|
TrueNAS 24.x (static inline array):
|
||||||
"filterByProviders",["STORJ_IX"]
|
"filterByProviders",["STORJ_IX"]
|
||||||
|
|
||||||
TrueNAS 25.x+ (Angular pureFunction binding):
|
TrueNAS 25.x+ (Angular pureFunction binding, inside a chained property call):
|
||||||
"filterByProviders",pe(115,Rn,i.CloudSyncProviderName.Storj)
|
c(2,"filterByProviders",pe(115,Rn,i.CloudSyncProviderName.Storj))("required",!0)
|
||||||
|
|
||||||
Both are replaced so the dropdown includes S3 and B2. The file is backed up
|
Both are replaced so the dropdown includes S3 and B2. The file is backed up
|
||||||
before modification so uninstall.sh can restore it.
|
before modification so uninstall.sh can restore it.
|
||||||
@@ -19,6 +19,7 @@ Safe to run multiple times — a marker string detects an already-patched file.
|
|||||||
Exits 0 in all cases (warnings are printed to stdout and logged by apply.sh).
|
Exits 0 in all cases (warnings are printed to stdout and logged by apply.sh).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
import contextlib
|
||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
import shutil
|
import shutil
|
||||||
@@ -32,11 +33,21 @@ WEBUI_CANDIDATES = [
|
|||||||
# Patterns tried in order; the first match wins.
|
# Patterns tried in order; the first match wins.
|
||||||
# Each entry is (compiled_regex, replacement_string).
|
# Each entry is (compiled_regex, replacement_string).
|
||||||
_PATTERNS = [
|
_PATTERNS = [
|
||||||
# TrueNAS 25.x+: Angular emits a pureFunction call instead of a literal array.
|
# TrueNAS 25.x+: Angular emits a pureFunction call instead of a literal array,
|
||||||
# pe / slot-index / factory-var / component-var are all minified and change
|
# inside a CHAINED property binding — so the call is followed by two closing
|
||||||
# across builds; CloudSyncProviderName.Storj is stable (TypeScript enum name).
|
# parens, one for pe(...) and one for the property(...) it sits in:
|
||||||
(re.compile(r'("filterByProviders",)\w+\(\d+,\w+,\w+\.CloudSyncProviderName\.Storj\)'),
|
#
|
||||||
r'\1["STORJ_IX","S3","B2"]'),
|
# c(2,"filterByProviders",pe(115,Rn,i.CloudSyncProviderName.Storj))("required",!0)
|
||||||
|
# ^^
|
||||||
|
# The pattern consumes both and re-emits one, leaving the paren balance
|
||||||
|
# unchanged. Getting that wrong is a syntax error in the bundle and the whole
|
||||||
|
# web UI goes blank — see tests/test_patch_ui.py.
|
||||||
|
#
|
||||||
|
# The minified names (pe / slot index / Rn / i) change across builds;
|
||||||
|
# CloudSyncProviderName.Storj is stable because it is a TypeScript enum name.
|
||||||
|
(re.compile(r'("filterByProviders",)\w+\(\d+,\w+,\w+\.CloudSyncProviderName\.Storj\)\)'),
|
||||||
|
r'\1["STORJ_IX","S3","B2"])'),
|
||||||
|
|
||||||
# TrueNAS 24.x and earlier: static inline array.
|
# TrueNAS 24.x and earlier: static inline array.
|
||||||
(re.compile(r'("filterByProviders",)\["STORJ_IX"\]'),
|
(re.compile(r'("filterByProviders",)\["STORJ_IX"\]'),
|
||||||
r'\1["STORJ_IX","S3","B2"]'),
|
r'\1["STORJ_IX","S3","B2"]'),
|
||||||
@@ -54,6 +65,11 @@ def _match_pattern(content):
|
|||||||
return None, None
|
return None, None
|
||||||
|
|
||||||
|
|
||||||
|
def _paren_delta(s):
|
||||||
|
"""Net parenthesis balance. Patching must not change it — see main()."""
|
||||||
|
return s.count("(") - s.count(")")
|
||||||
|
|
||||||
|
|
||||||
def find_bundle():
|
def find_bundle():
|
||||||
"""
|
"""
|
||||||
Search WEBUI_CANDIDATES for the JS chunk containing the filterByProviders
|
Search WEBUI_CANDIDATES for the JS chunk containing the filterByProviders
|
||||||
@@ -124,6 +140,24 @@ def main():
|
|||||||
)
|
)
|
||||||
return
|
return
|
||||||
|
|
||||||
|
# Never write JS whose parentheses we have unbalanced. A pattern that eats one
|
||||||
|
# paren too many is a syntax error in the bundle and the entire TrueNAS web UI
|
||||||
|
# goes blank -- and because MARKER is then present, every later run reports
|
||||||
|
# "already patched" and skips, so the patch cannot heal itself. Recovery means
|
||||||
|
# hand-restoring the .pre-truecloud-patch backup.
|
||||||
|
#
|
||||||
|
# This is not hypothetical: it shipped once. Refuse instead.
|
||||||
|
if _paren_delta(patched) != _paren_delta(content):
|
||||||
|
print(
|
||||||
|
"[truecloud-patch] ERROR: the replacement would unbalance the bundle's "
|
||||||
|
"parentheses — refusing to write.\n"
|
||||||
|
"[truecloud-patch] The UI is UNCHANGED and still works. This means the "
|
||||||
|
"pattern no longer fits this TrueNAS build.\n"
|
||||||
|
"[truecloud-patch] File an issue at "
|
||||||
|
"https://github.com/sudolulo/truenas-truecloud-patch"
|
||||||
|
)
|
||||||
|
return
|
||||||
|
|
||||||
tmp = path + ".tmp"
|
tmp = path + ".tmp"
|
||||||
try:
|
try:
|
||||||
with open(tmp, "w", encoding="utf-8") as fh:
|
with open(tmp, "w", encoding="utf-8") as fh:
|
||||||
@@ -131,10 +165,8 @@ def main():
|
|||||||
os.replace(tmp, path)
|
os.replace(tmp, path)
|
||||||
except OSError as exc:
|
except OSError as exc:
|
||||||
print(f"[truecloud-patch] ERROR: Could not write {path}: {exc}")
|
print(f"[truecloud-patch] ERROR: Could not write {path}: {exc}")
|
||||||
try:
|
with contextlib.suppress(OSError):
|
||||||
os.unlink(tmp)
|
os.unlink(tmp)
|
||||||
except OSError:
|
|
||||||
pass
|
|
||||||
return
|
return
|
||||||
|
|
||||||
print(f"[truecloud-patch] UI bundle patched ({count} replacement(s)): {path}")
|
print(f"[truecloud-patch] UI bundle patched ({count} replacement(s)): {path}")
|
||||||
|
|||||||
@@ -0,0 +1,565 @@
|
|||||||
|
"""Nested-dataset snapshot support for TrueCloud Backup.
|
||||||
|
|
||||||
|
Why this exists
|
||||||
|
---------------
|
||||||
|
Stock TrueNAS refuses ``snapshot = true`` when the backup path contains child
|
||||||
|
datasets::
|
||||||
|
|
||||||
|
This option is only available for datasets that have no further nesting
|
||||||
|
|
||||||
|
That guard is *correct* and it is not laziness. ``plugins/cloud/snapshot.py``
|
||||||
|
already takes a **recursive** ZFS snapshot, but it then points the backup tool
|
||||||
|
at the *parent* dataset's ``.zfs/snapshot/<snap>/`` directory -- and ZFS does
|
||||||
|
not expose child datasets through a parent's snapshot directory::
|
||||||
|
|
||||||
|
/mnt/Tap/.zfs/snapshot/<snap>/apps/ -> 0 entries
|
||||||
|
/mnt/Tap/apps/lidarr/config/.zfs/snapshot/<snap>/ -> the real data
|
||||||
|
|
||||||
|
So without the guard, the backup tool would walk a near-empty tree, report
|
||||||
|
SUCCESS, and upload almost nothing. A backup that lies about succeeding is the
|
||||||
|
worst failure a backup system can have, so middleware refuses the config
|
||||||
|
instead.
|
||||||
|
|
||||||
|
This module implements the missing half: after the (already recursive) snapshot
|
||||||
|
is taken, every descendant dataset's *own* ``.zfs/snapshot/<snap>`` directory is
|
||||||
|
bind-mounted into a staging tree that mirrors the original layout. The backup
|
||||||
|
tool is then pointed at the staging root, which is a complete, consistent,
|
||||||
|
point-in-time view of the whole subtree.
|
||||||
|
|
||||||
|
Cardinal safety rule
|
||||||
|
--------------------
|
||||||
|
**If the tree cannot be staged completely, fail loudly.** Never return a partial
|
||||||
|
tree. Silently backing up an incomplete tree is precisely the failure this
|
||||||
|
feature exists to prevent, and it would be worse than not having the feature.
|
||||||
|
|
||||||
|
Snapshot lifecycle -- read this before changing anything
|
||||||
|
--------------------------------------------------------
|
||||||
|
``zfs.snapshot.delete`` defaults to ``recursive=False``, and stock
|
||||||
|
``restic_backup()`` calls it with no options. Stock gets away with that because
|
||||||
|
its validation means ``recursive`` is never actually True in the field. Enabling
|
||||||
|
nested datasets makes recursive snapshots real, so the parent
|
||||||
|
(``Tap@snap``) has one child snapshot per descendant dataset (160+ here).
|
||||||
|
Deleting only the parent would orphan every child on **every successful run**.
|
||||||
|
|
||||||
|
Therefore this module owns the whole lifecycle:
|
||||||
|
|
||||||
|
* :func:`delete_snapshot_tree` sweeps the parent *and* every child snapshot, and
|
||||||
|
is idempotent -- it copes with stock's ``finally`` having already removed the
|
||||||
|
parent.
|
||||||
|
* The snapshot name is recorded in a sidecar file next to the staging root, not
|
||||||
|
only in memory, so a middlewared restart mid-backup cannot orphan it.
|
||||||
|
* Bind-mounting ``.zfs/snapshot/<snap>`` pins the snapshot, so stock's delete
|
||||||
|
fails with EBUSY and logs one benign warning; we unmount and then sweep.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import contextlib
|
||||||
|
import os
|
||||||
|
import stat
|
||||||
|
import subprocess
|
||||||
|
|
||||||
|
__all__ = [
|
||||||
|
"STAGING_BASE",
|
||||||
|
"StagingError",
|
||||||
|
"apply_plan",
|
||||||
|
"cleanup_all",
|
||||||
|
"cleanup_task",
|
||||||
|
"current_mounts_under",
|
||||||
|
"delete_snapshot_tree",
|
||||||
|
"plan_staging",
|
||||||
|
"sidecar_for",
|
||||||
|
"snapshot_tree_names",
|
||||||
|
"stage_nested",
|
||||||
|
"staging_root_for",
|
||||||
|
"teardown",
|
||||||
|
"verify_staged",
|
||||||
|
]
|
||||||
|
|
||||||
|
#: Where staging trees are assembled. tmpfs; bind mounts consume no space.
|
||||||
|
STAGING_BASE = "/run/truecloud-nested"
|
||||||
|
|
||||||
|
# Which snapshot a staging tree pins is recorded ONLY in the sidecar file, never
|
||||||
|
# also in memory. An in-process dict would be a second source of truth that a
|
||||||
|
# middlewared restart silently empties -- and it is exactly the restart case that
|
||||||
|
# must not orphan a 250-snapshot tree. One record, on disk, or none.
|
||||||
|
|
||||||
|
|
||||||
|
class StagingError(Exception):
|
||||||
|
"""Staging could not produce a complete tree. The backup must not proceed."""
|
||||||
|
|
||||||
|
|
||||||
|
# ── pure helpers ──────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def staging_root_for(name: str, base: str | None = None) -> str:
|
||||||
|
"""Stable staging root for a task name (e.g. ``cloud_backup-5``).
|
||||||
|
|
||||||
|
``base`` defaults to :data:`STAGING_BASE` at CALL time, not at import time --
|
||||||
|
a ``base=STAGING_BASE`` default would freeze the value into the function
|
||||||
|
object and silently ignore any later override.
|
||||||
|
"""
|
||||||
|
if base is None:
|
||||||
|
base = STAGING_BASE
|
||||||
|
safe = "".join(c if (c.isalnum() or c in "-_.") else "_" for c in name)
|
||||||
|
# A component of "." or ".." would escape STAGING_BASE once joined.
|
||||||
|
if not safe or safe.strip(".") == "":
|
||||||
|
safe = "task"
|
||||||
|
return os.path.join(base, safe)
|
||||||
|
|
||||||
|
|
||||||
|
def sidecar_for(staging_root: str) -> str:
|
||||||
|
"""Path of the file recording which ZFS snapshot a staging tree pins."""
|
||||||
|
return staging_root + ".snapshot"
|
||||||
|
|
||||||
|
|
||||||
|
def _write_sidecar(staging_root: str, snapshot: str) -> None:
|
||||||
|
"""Record the pinned snapshot on disk. Blocking; call via run_in_thread."""
|
||||||
|
with contextlib.suppress(OSError):
|
||||||
|
os.makedirs(os.path.dirname(staging_root), exist_ok=True)
|
||||||
|
with open(sidecar_for(staging_root), "w", encoding="utf-8") as fh:
|
||||||
|
fh.write(snapshot)
|
||||||
|
|
||||||
|
|
||||||
|
def _read_sidecar(staging_root: str) -> str | None:
|
||||||
|
"""The snapshot a previous run recorded here, if any."""
|
||||||
|
try:
|
||||||
|
with open(sidecar_for(staging_root), encoding="utf-8") as fh:
|
||||||
|
return fh.read().strip() or None
|
||||||
|
except OSError:
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _remove_sidecar(staging_root: str) -> None:
|
||||||
|
with contextlib.suppress(OSError):
|
||||||
|
os.unlink(sidecar_for(staging_root))
|
||||||
|
|
||||||
|
|
||||||
|
def _depth(path: str) -> int:
|
||||||
|
return len([p for p in path.split("/") if p])
|
||||||
|
|
||||||
|
|
||||||
|
def snapshot_tree_names(snapshot: str, all_names) -> list[str]:
|
||||||
|
"""Every snapshot produced by ``zfs snapshot -r <dataset>@<snap>``.
|
||||||
|
|
||||||
|
That is the parent plus one per descendant dataset, all sharing the same
|
||||||
|
name after the ``@``. Pure, so the sweep logic is testable without ZFS.
|
||||||
|
"""
|
||||||
|
dataset, _, snapname = snapshot.partition("@")
|
||||||
|
if not snapname:
|
||||||
|
return []
|
||||||
|
parent = f"{dataset}@{snapname}"
|
||||||
|
prefix = dataset + "/"
|
||||||
|
suffix = "@" + snapname
|
||||||
|
return [
|
||||||
|
n for n in all_names
|
||||||
|
if n == parent or (n.startswith(prefix) and n.endswith(suffix))
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def _probe_snapdir(path):
|
||||||
|
"""Classify a snapshot directory: ``ok``, ``missing``, or why it is unusable.
|
||||||
|
|
||||||
|
``os.path.isdir()`` collapses "does not exist" and "cannot stat" into the
|
||||||
|
same ``False``, so an EACCES would report itself as "has no snapshot" and
|
||||||
|
send someone hunting for a snapshot that is sitting right there. Both cases
|
||||||
|
still abort the backup -- but it has to say which one.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
st = os.stat(path)
|
||||||
|
except FileNotFoundError:
|
||||||
|
return "missing"
|
||||||
|
except OSError as e:
|
||||||
|
return f"cannot be read ({e.strerror})"
|
||||||
|
return "ok" if stat.S_ISDIR(st.st_mode) else "is not a directory"
|
||||||
|
|
||||||
|
|
||||||
|
def plan_staging(base_dataset, base_mountpoint, path, snapshot_name, datasets,
|
||||||
|
staging_root, probe=_probe_snapdir):
|
||||||
|
"""Compute the bind-mount plan for staging a nested tree. Pure function.
|
||||||
|
|
||||||
|
``datasets`` is a list of dicts shaped like ``zfs.dataset.query`` results:
|
||||||
|
``{"name": str, "properties": {"mountpoint": {"value": str},
|
||||||
|
"mounted": {"value": "yes"|"no"}}}``.
|
||||||
|
|
||||||
|
Returns ``(mounts, skipped)`` where ``mounts`` is an ordered list of
|
||||||
|
``(source, target)`` pairs (parents before children) and ``skipped`` is a
|
||||||
|
list of ``(dataset_name, reason)`` covering only datasets that are *in
|
||||||
|
scope* -- i.e. descendants of ``base_dataset``. Datasets elsewhere on the
|
||||||
|
system are ignored silently; reporting them would bury the ones that matter.
|
||||||
|
|
||||||
|
Raises StagingError if an in-scope descendant holds data we would otherwise
|
||||||
|
silently omit.
|
||||||
|
"""
|
||||||
|
def snapdir(mountpoint):
|
||||||
|
return os.path.join(mountpoint, ".zfs", "snapshot", snapshot_name)
|
||||||
|
|
||||||
|
# Root of the staging tree: the backup path as seen inside the base
|
||||||
|
# dataset's own snapshot.
|
||||||
|
rel = os.path.relpath(path, base_mountpoint)
|
||||||
|
root_src = snapdir(base_mountpoint)
|
||||||
|
if rel != ".":
|
||||||
|
root_src = os.path.join(root_src, rel)
|
||||||
|
|
||||||
|
mounts = [(root_src, staging_root)]
|
||||||
|
skipped = []
|
||||||
|
|
||||||
|
ds_prefix = base_dataset.rstrip("/") + "/"
|
||||||
|
path_prefix = path.rstrip("/") + "/"
|
||||||
|
|
||||||
|
for ds in datasets:
|
||||||
|
name = ds.get("name", "")
|
||||||
|
# Scope by DATASET NAME, not mountpoint: a dataset with no mountpoint
|
||||||
|
# cannot be scoped by path, and scoping by path first would drag in
|
||||||
|
# every mountpoint-less dataset on the box (all of Tank/.system/*, ...).
|
||||||
|
if not name.startswith(ds_prefix):
|
||||||
|
continue
|
||||||
|
|
||||||
|
props = ds.get("properties", {})
|
||||||
|
mp = props.get("mountpoint", {}).get("value", "")
|
||||||
|
|
||||||
|
if not mp or mp in ("none", "legacy", "-"):
|
||||||
|
skipped.append((name, f"mountpoint is {mp or 'unset'}"))
|
||||||
|
continue
|
||||||
|
|
||||||
|
if not mp.startswith(path_prefix):
|
||||||
|
# A descendant dataset mounted outside the backed-up path is
|
||||||
|
# genuinely not part of this tree. Not an omission.
|
||||||
|
continue
|
||||||
|
|
||||||
|
if props.get("mounted", {}).get("value", "yes") == "no":
|
||||||
|
# An unmounted (e.g. locked/encrypted) dataset contributes nothing to
|
||||||
|
# the live tree either, so skipping matches stock semantics -- but it
|
||||||
|
# is a real gap and must be visible, never silent.
|
||||||
|
skipped.append((name, "dataset is not mounted (locked/encrypted?)"))
|
||||||
|
continue
|
||||||
|
|
||||||
|
src = snapdir(mp)
|
||||||
|
status = probe(src)
|
||||||
|
if status != "ok":
|
||||||
|
# Either the recursive snapshot missed this dataset, or we cannot read
|
||||||
|
# it. Either way its data would be silently omitted. Refuse -- but say
|
||||||
|
# WHICH, because "no snapshot" and "permission denied" send you to
|
||||||
|
# completely different places.
|
||||||
|
detail = (
|
||||||
|
f"has no snapshot {snapshot_name!r}" if status == "missing"
|
||||||
|
else f"snapshot {snapshot_name!r} {status}"
|
||||||
|
)
|
||||||
|
raise StagingError(
|
||||||
|
f"dataset {name!r} {detail} at {src!r}; "
|
||||||
|
f"refusing to back up an incomplete tree"
|
||||||
|
)
|
||||||
|
|
||||||
|
mounts.append((src, os.path.join(staging_root, os.path.relpath(mp, path))))
|
||||||
|
|
||||||
|
# Parents before children, so each mountpoint exists before we mount onto it.
|
||||||
|
mounts.sort(key=lambda m: _depth(m[1]))
|
||||||
|
return mounts, skipped
|
||||||
|
|
||||||
|
|
||||||
|
def current_mounts_under(root, mounts_file="/proc/self/mounts"):
|
||||||
|
"""Mountpoints at or under ``root``, deepest first. Used for teardown."""
|
||||||
|
found = []
|
||||||
|
try:
|
||||||
|
with open(mounts_file, encoding="utf-8") as fh:
|
||||||
|
for line in fh:
|
||||||
|
parts = line.split()
|
||||||
|
if len(parts) < 2:
|
||||||
|
continue
|
||||||
|
mp = parts[1].replace("\\040", " ").replace("\\011", "\t")
|
||||||
|
if mp == root or mp.startswith(root.rstrip("/") + "/"):
|
||||||
|
found.append(mp)
|
||||||
|
except OSError:
|
||||||
|
return []
|
||||||
|
found.sort(key=_depth, reverse=True)
|
||||||
|
return found
|
||||||
|
|
||||||
|
|
||||||
|
# ── mount / unmount ───────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def _run(cmd):
|
||||||
|
return subprocess.run(cmd, capture_output=True, text=True, check=False)
|
||||||
|
|
||||||
|
|
||||||
|
def apply_plan(mounts, runner=_run, isdir=os.path.isdir):
|
||||||
|
"""Execute the bind-mount plan. Blocking; call via ``run_in_thread``.
|
||||||
|
|
||||||
|
Raises StagingError on the first failure, after rolling back what was
|
||||||
|
mounted -- a half-built tree must never be handed to the backup tool.
|
||||||
|
"""
|
||||||
|
if not mounts:
|
||||||
|
raise StagingError("empty staging plan")
|
||||||
|
|
||||||
|
staging_root = mounts[0][1]
|
||||||
|
done = []
|
||||||
|
try:
|
||||||
|
os.makedirs(staging_root, exist_ok=True)
|
||||||
|
for src, target in mounts:
|
||||||
|
if not isdir(target):
|
||||||
|
# Child mountpoint dirs come from the parent snapshot, which is
|
||||||
|
# read-only -- we cannot mkdir them. Only the root is ours.
|
||||||
|
raise StagingError(f"staging target {target!r} does not exist")
|
||||||
|
res = runner(["mount", "--bind", src, target])
|
||||||
|
if res.returncode != 0:
|
||||||
|
raise StagingError(
|
||||||
|
f"bind-mount {src!r} -> {target!r} failed: "
|
||||||
|
f"{(res.stderr or '').strip() or res.returncode}"
|
||||||
|
)
|
||||||
|
done.append(target)
|
||||||
|
except Exception:
|
||||||
|
for target in reversed(done):
|
||||||
|
runner(["umount", "-l", target])
|
||||||
|
with contextlib.suppress(OSError):
|
||||||
|
os.rmdir(staging_root)
|
||||||
|
raise
|
||||||
|
return staging_root
|
||||||
|
|
||||||
|
|
||||||
|
def verify_staged(mounts, ismount=os.path.ismount, listdir=os.listdir):
|
||||||
|
"""Assert the staged tree is real and complete. Raises StagingError if not.
|
||||||
|
|
||||||
|
This is the anti-regression guard: it is what stops this feature from ever
|
||||||
|
degrading back into the silently-empty backup that the stock validation
|
||||||
|
refuses to allow.
|
||||||
|
"""
|
||||||
|
if not mounts:
|
||||||
|
raise StagingError("nothing was staged")
|
||||||
|
|
||||||
|
staging_root = mounts[0][1]
|
||||||
|
for _src, target in mounts:
|
||||||
|
if not ismount(target):
|
||||||
|
raise StagingError(f"staging target {target!r} is not a mountpoint")
|
||||||
|
|
||||||
|
try:
|
||||||
|
if not listdir(staging_root):
|
||||||
|
raise StagingError(f"staging root {staging_root!r} is empty")
|
||||||
|
except OSError as e:
|
||||||
|
raise StagingError(f"staging root {staging_root!r} unreadable: {e}") from e
|
||||||
|
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def teardown(staging_root, runner=_run, mounts_file="/proc/self/mounts"):
|
||||||
|
"""Unmount the staging tree (deepest first) and remove the root.
|
||||||
|
|
||||||
|
Idempotent, and does not depend on an in-memory plan -- so it also cleans up
|
||||||
|
leftovers from a crashed run.
|
||||||
|
"""
|
||||||
|
errors = []
|
||||||
|
for mp in current_mounts_under(staging_root, mounts_file=mounts_file):
|
||||||
|
res = runner(["umount", mp])
|
||||||
|
if res.returncode != 0:
|
||||||
|
res = runner(["umount", "-l", mp]) # lazy: better than leaking
|
||||||
|
if res.returncode != 0:
|
||||||
|
errors.append(f"{mp}: {(res.stderr or '').strip()}")
|
||||||
|
with contextlib.suppress(OSError):
|
||||||
|
os.rmdir(staging_root)
|
||||||
|
return errors
|
||||||
|
|
||||||
|
|
||||||
|
# ── async orchestration (middleware is duck-typed; no middlewared import) ─────
|
||||||
|
|
||||||
|
|
||||||
|
async def delete_snapshot_tree(middleware, snapshot, logger=None):
|
||||||
|
"""Delete the parent snapshot AND every child created by ``zfs snapshot -r``.
|
||||||
|
|
||||||
|
``zfs.snapshot.delete`` is non-recursive by default and stock calls it with
|
||||||
|
no options, so relying on stock would orphan one snapshot per descendant
|
||||||
|
dataset on every run. Idempotent: tolerates the parent already being gone
|
||||||
|
(stock's ``finally`` may have won the race once our mounts were released).
|
||||||
|
"""
|
||||||
|
dataset = snapshot.partition("@")[0]
|
||||||
|
|
||||||
|
# Fast path: ONE recursive delete removes the parent and every child that
|
||||||
|
# `zfs snapshot -r` created (252 on a real pool). Deleting them individually
|
||||||
|
# also works, but it is neither cheap nor atomic -- a run killed part-way
|
||||||
|
# through 252 sequential deletes leaves exactly the orphans this function
|
||||||
|
# exists to prevent.
|
||||||
|
try:
|
||||||
|
await middleware.call("zfs.snapshot.delete", snapshot, {"recursive": True})
|
||||||
|
return
|
||||||
|
except Exception: # noqa: BLE001 - fall through to the explicit sweep
|
||||||
|
pass
|
||||||
|
|
||||||
|
# The parent may already be gone -- stock's `finally` can win the race once
|
||||||
|
# our mounts are released -- which fails the recursive delete while the
|
||||||
|
# children survive. Sweep them by name.
|
||||||
|
try:
|
||||||
|
snaps = await middleware.call(
|
||||||
|
"zfs.snapshot.query", [["name", "^", dataset]], {"select": ["name"]}
|
||||||
|
)
|
||||||
|
# An empty result means the tree is already gone -- delete nothing, and
|
||||||
|
# do not fall back to the parent, which would only log a spurious
|
||||||
|
# "does not exist" warning on every clean run.
|
||||||
|
names = snapshot_tree_names(snapshot, [s["name"] for s in snaps])
|
||||||
|
except Exception as e: # noqa: BLE001 - fall back to at least the parent
|
||||||
|
if logger:
|
||||||
|
logger.warning(
|
||||||
|
"truecloud-patch: could not enumerate snapshot tree for %s: %r",
|
||||||
|
snapshot, e,
|
||||||
|
)
|
||||||
|
names = [snapshot]
|
||||||
|
|
||||||
|
for name in names:
|
||||||
|
try:
|
||||||
|
await middleware.call("zfs.snapshot.delete", name)
|
||||||
|
except Exception as e: # noqa: BLE001 - already gone is fine
|
||||||
|
if logger:
|
||||||
|
logger.warning(
|
||||||
|
"truecloud-patch: could not delete snapshot %s: %r", name, e
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def stage_nested(middleware, path, snapshot, base_dataset, base_mountpoint,
|
||||||
|
task_name, datasets, logger=None):
|
||||||
|
"""Build a complete staging tree for `path` from the already-taken `snapshot`.
|
||||||
|
|
||||||
|
`snapshot` is a full ZFS snapshot name ("Tap@cloud_backup-5-2026...").
|
||||||
|
|
||||||
|
`datasets` is the FILESYSTEM dataset list. **It MUST have been enumerated
|
||||||
|
AFTER `snapshot` was taken.** A list read beforehand can miss a dataset
|
||||||
|
created in the gap: the recursive snapshot would capture it, but the staging
|
||||||
|
plan would not, and its data would be silently omitted from the backup.
|
||||||
|
Enumerated afterwards, an unsnapshotted dataset instead trips the isdir()
|
||||||
|
check in plan_staging and fails the run loudly.
|
||||||
|
|
||||||
|
Returns the staging root to hand to the backup tool.
|
||||||
|
|
||||||
|
Raises StagingError if the tree cannot be staged completely -- the caller
|
||||||
|
must let that propagate so the backup fails instead of silently uploading a
|
||||||
|
partial tree. The caller is responsible for deleting `snapshot` in that case
|
||||||
|
(see SNAPSHOT_BLOCK in apply.sh).
|
||||||
|
"""
|
||||||
|
snapshot_name = snapshot.split("@", 1)[1]
|
||||||
|
staging_root = staging_root_for(task_name)
|
||||||
|
|
||||||
|
# A previous run may have crashed mid-flight; never build on top of that.
|
||||||
|
await middleware.run_in_thread(teardown, staging_root)
|
||||||
|
|
||||||
|
# ...and if it left a sidecar behind, that snapshot tree is still on disk and
|
||||||
|
# nothing else will ever reclaim it. Sweep it before we overwrite the record,
|
||||||
|
# or a single crashed run orphans 160+ snapshots permanently.
|
||||||
|
stale = await middleware.run_in_thread(_read_sidecar, staging_root)
|
||||||
|
if stale and stale != snapshot:
|
||||||
|
if logger:
|
||||||
|
logger.warning(
|
||||||
|
"truecloud-patch: reclaiming snapshot tree from an earlier "
|
||||||
|
"interrupted run: %s", stale,
|
||||||
|
)
|
||||||
|
await delete_snapshot_tree(middleware, stale, logger=logger)
|
||||||
|
|
||||||
|
# Record the snapshot BEFORE mounting anything, not after. middlewared can
|
||||||
|
# die at any point (this patch even schedules a restart at boot), and the
|
||||||
|
# sidecar is the only thing that survives it -- an in-process dict would take
|
||||||
|
# the sole record of a 160-snapshot tree with it. Writing it after apply_plan
|
||||||
|
# would leave exactly the crash window the sidecar exists to close.
|
||||||
|
await middleware.run_in_thread(_write_sidecar, staging_root, snapshot)
|
||||||
|
|
||||||
|
try:
|
||||||
|
mounts, skipped = await middleware.run_in_thread(
|
||||||
|
plan_staging, base_dataset, base_mountpoint, path, snapshot_name,
|
||||||
|
datasets, staging_root,
|
||||||
|
)
|
||||||
|
if logger:
|
||||||
|
for name, reason in skipped:
|
||||||
|
logger.warning(
|
||||||
|
"truecloud-patch: not staging dataset %r: %s", name, reason
|
||||||
|
)
|
||||||
|
|
||||||
|
await middleware.run_in_thread(apply_plan, mounts)
|
||||||
|
await middleware.run_in_thread(verify_staged, mounts)
|
||||||
|
except Exception:
|
||||||
|
await middleware.run_in_thread(teardown, staging_root)
|
||||||
|
await middleware.run_in_thread(_remove_sidecar, staging_root)
|
||||||
|
raise
|
||||||
|
|
||||||
|
if logger:
|
||||||
|
logger.info(
|
||||||
|
"truecloud-patch: staged %d dataset(s) from %s at %s",
|
||||||
|
len(mounts), snapshot, staging_root,
|
||||||
|
)
|
||||||
|
return staging_root
|
||||||
|
|
||||||
|
|
||||||
|
async def cleanup_task(middleware, task_name, logger=None):
|
||||||
|
"""Tear down a task's staging tree and delete the snapshot it pinned.
|
||||||
|
|
||||||
|
Safe to call unconditionally: a no-op when the task was never staged.
|
||||||
|
"""
|
||||||
|
staging_root = staging_root_for(task_name)
|
||||||
|
snapshot = _read_sidecar(staging_root)
|
||||||
|
|
||||||
|
if snapshot is None and not os.path.isdir(staging_root):
|
||||||
|
return # never staged; nothing to do
|
||||||
|
|
||||||
|
errors = await middleware.run_in_thread(teardown, staging_root)
|
||||||
|
if errors and logger:
|
||||||
|
for err in errors:
|
||||||
|
logger.warning("truecloud-patch: staging teardown: %s", err)
|
||||||
|
|
||||||
|
if snapshot is not None:
|
||||||
|
await delete_snapshot_tree(middleware, snapshot, logger=logger)
|
||||||
|
|
||||||
|
_remove_sidecar(staging_root)
|
||||||
|
|
||||||
|
|
||||||
|
# ── offline cleanup (uninstall.sh / recover.sh) ───────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def cleanup_all(base=None, runner=_run, mounts_file="/proc/self/mounts",
|
||||||
|
glob_fn=None, read_sidecar=_read_sidecar):
|
||||||
|
"""Tear down every staging tree. Used by uninstall.sh and recover.sh.
|
||||||
|
|
||||||
|
Those scripts must work when middlewared is dead, so they cannot go through
|
||||||
|
the async path -- but they must not reimplement the teardown either: the
|
||||||
|
depth-ordering and lazy-umount fallback are fiddly, and a second copy in
|
||||||
|
shell would be the untested one. This is the same tested code.
|
||||||
|
|
||||||
|
Returns ``(lines, errors)``: report lines to print, and unmount errors.
|
||||||
|
"""
|
||||||
|
import glob as _glob
|
||||||
|
|
||||||
|
base = base or STAGING_BASE
|
||||||
|
glob_fn = glob_fn or _glob.glob
|
||||||
|
lines = []
|
||||||
|
|
||||||
|
# Report orphaned snapshots BEFORE removing the sidecars that name them --
|
||||||
|
# a sidecar is the only record that an interrupted run's snapshot tree (one
|
||||||
|
# snapshot per descendant dataset) is still on disk.
|
||||||
|
for sc in sorted(glob_fn(os.path.join(base, "*.snapshot"))):
|
||||||
|
snap = read_sidecar(sc[: -len(".snapshot")])
|
||||||
|
if snap:
|
||||||
|
lines.append(f" NOTE: an interrupted backup left snapshot '{snap}' behind.")
|
||||||
|
lines.append(f" Remove it and its children: zfs destroy -r '{snap}'")
|
||||||
|
|
||||||
|
mounts = current_mounts_under(base, mounts_file=mounts_file)
|
||||||
|
if not mounts:
|
||||||
|
lines.append(" None active.")
|
||||||
|
for mp in mounts:
|
||||||
|
lines.append(f" Unmounting: {mp}")
|
||||||
|
|
||||||
|
errors = teardown(base, runner=runner, mounts_file=mounts_file)
|
||||||
|
for err in errors:
|
||||||
|
lines.append(f" WARNING: could not unmount {err}")
|
||||||
|
|
||||||
|
if not errors:
|
||||||
|
for sc in glob_fn(os.path.join(base, "*.snapshot")):
|
||||||
|
with contextlib.suppress(OSError):
|
||||||
|
os.unlink(sc)
|
||||||
|
with contextlib.suppress(OSError):
|
||||||
|
os.rmdir(base)
|
||||||
|
|
||||||
|
return lines, errors
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
import sys
|
||||||
|
|
||||||
|
if len(sys.argv) > 1 and sys.argv[1] == "cleanup":
|
||||||
|
_lines, _errors = cleanup_all()
|
||||||
|
for _line in _lines:
|
||||||
|
print(_line)
|
||||||
|
sys.exit(1 if _errors else 0)
|
||||||
|
print("usage: truecloud_nested.py cleanup", file=sys.stderr)
|
||||||
|
sys.exit(2)
|
||||||
Executable
+55
@@ -0,0 +1,55 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
# patch/wait_restart.sh — payload of the transient `truecloud-mw-restart`
|
||||||
|
# unit that apply.sh schedules in boot context (Step 3).
|
||||||
|
#
|
||||||
|
# Why not restart middlewared directly from the unit: systemd ordering
|
||||||
|
# (`After=multi-user.target`, used up to v0.0.4) cannot see middlewared's
|
||||||
|
# *internal* boot work. When the boot targets are reached, two things are
|
||||||
|
# typically still in flight inside middlewared:
|
||||||
|
#
|
||||||
|
# - ix-reporting.service's `midclt call reporting.start_service` (netdata,
|
||||||
|
# which feeds the dashboard hardware stats), and
|
||||||
|
# - the docker/apps startup task middlewared creates on its own
|
||||||
|
# system-ready event (`docker.state.start_service`).
|
||||||
|
#
|
||||||
|
# Restarting middlewared while those run kills them, and nothing retries
|
||||||
|
# them until the next boot: every app stays down (`docker.status` FAILED),
|
||||||
|
# the dashboard shows no stats, and middleware-internal service state (e.g.
|
||||||
|
# the SMB backend) is left uninitialized. Observed on 25.10.4 with v0.0.4.
|
||||||
|
#
|
||||||
|
# So this script waits for both layers to settle before restarting. Every
|
||||||
|
# wait is bounded and fails open: worst case the restart still happens, just
|
||||||
|
# later — a restart on a settled system is harmless (docker, apps and
|
||||||
|
# netdata are independent processes; only the middleware API blips).
|
||||||
|
#
|
||||||
|
# NOTE: the unit must NOT be Type=oneshot. A oneshot's start job stays in
|
||||||
|
# the systemd job queue until the process exits, and `is-system-running
|
||||||
|
# --wait` below waits for that same queue to drain — the unit would deadlock
|
||||||
|
# on itself until the timeout. apply.sh schedules this with the default
|
||||||
|
# service type, whose start job completes at fork.
|
||||||
|
|
||||||
|
# 1. systemd layer: wait for the boot job queue to drain. This covers every
|
||||||
|
# ix-* oneshot still activating, including ix-reporting's in-flight midclt
|
||||||
|
# call. The exit code is irrelevant — a "degraded" boot (any unrelated
|
||||||
|
# failed unit) is still a finished boot. The timeout only guards against
|
||||||
|
# a boot that never settles (e.g. a unit stuck on a network wait).
|
||||||
|
timeout 900 systemctl is-system-running --wait > /dev/null 2>&1
|
||||||
|
|
||||||
|
# 2. middlewared layer: poll the docker state machine until it leaves the
|
||||||
|
# transitional states (PENDING/INITIALIZING/STOPPING/MIGRATING — see
|
||||||
|
# middlewared/plugins/docker/state_utils.py). An empty answer means
|
||||||
|
# midclt could not respond at all; keep waiting. Cap at 10 minutes.
|
||||||
|
for _ in $(seq 1 120); do
|
||||||
|
_status=$(midclt call docker.status 2>/dev/null \
|
||||||
|
| grep -oE '"status": "[A-Z_]+"' | cut -d'"' -f4)
|
||||||
|
case "$_status" in
|
||||||
|
RUNNING|STOPPED|UNCONFIGURED|FAILED|MIGRATION_FAILED) break ;;
|
||||||
|
esac
|
||||||
|
sleep 5
|
||||||
|
done
|
||||||
|
|
||||||
|
# 3. Grace period for middleware-internal ready-event tasks that expose no
|
||||||
|
# queryable state (smb.configure and friends). Bounded insurance.
|
||||||
|
sleep 30
|
||||||
|
|
||||||
|
exec systemctl try-restart middlewared
|
||||||
@@ -0,0 +1,15 @@
|
|||||||
|
# Tooling config only — this project is not a Python package. The patch modules
|
||||||
|
# are copied into middlewared's site-packages by patch/apply.sh at boot.
|
||||||
|
|
||||||
|
[tool.ruff]
|
||||||
|
line-length = 100
|
||||||
|
target-version = "py311"
|
||||||
|
|
||||||
|
[tool.ruff.lint]
|
||||||
|
select = ["E", "F", "W", "B", "UP", "SIM"]
|
||||||
|
ignore = [
|
||||||
|
"E501", # long lines in explanatory comments are fine
|
||||||
|
]
|
||||||
|
|
||||||
|
[tool.pytest.ini_options]
|
||||||
|
testpaths = ["tests"]
|
||||||
+10
-1
@@ -17,10 +17,11 @@
|
|||||||
# bash /mnt/tank/truenas-truecloud-patch/patch/apply.sh
|
# bash /mnt/tank/truenas-truecloud-patch/patch/apply.sh
|
||||||
# systemctl restart middlewared
|
# systemctl restart middlewared
|
||||||
|
|
||||||
VERSION="0.0.4"
|
VERSION="0.3.2"
|
||||||
|
|
||||||
PATCH_DIR="$(cd "$(dirname "$0")" && pwd)"
|
PATCH_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||||
|
|
||||||
|
|
||||||
echo "=== TrueNAS TrueCloud Provider Patch v${VERSION} — Recover ==="
|
echo "=== TrueNAS TrueCloud Provider Patch v${VERSION} — Recover ==="
|
||||||
echo ""
|
echo ""
|
||||||
|
|
||||||
@@ -52,6 +53,14 @@ for _tag in mw ui; do
|
|||||||
done
|
done
|
||||||
[ "$_any" -eq 0 ] && echo " No overlays active."
|
[ "$_any" -eq 0 ] && echo " No overlays active."
|
||||||
|
|
||||||
|
# Nested-snapshot staging trees are bind mounts that PIN their ZFS snapshots, so
|
||||||
|
# leaving them mounted blocks those snapshots from ever being destroyed. The
|
||||||
|
# overlays above are volatile, but these are not self-healing without a reboot,
|
||||||
|
# and recover.sh is expected to work without one.
|
||||||
|
echo "Unmounting nested-snapshot staging trees ..."
|
||||||
|
# Best-effort: never block recovery. Same tested implementation as uninstall.sh.
|
||||||
|
python3 "$PATCH_DIR/patch/truecloud_nested.py" cleanup || true
|
||||||
|
|
||||||
# Cancel a deferred boot restart if one is still queued — we restart ourselves.
|
# Cancel a deferred boot restart if one is still queued — we restart ourselves.
|
||||||
systemctl stop truecloud-mw-restart.service 2>/dev/null
|
systemctl stop truecloud-mw-restart.service 2>/dev/null
|
||||||
systemctl reset-failed truecloud-mw-restart.service 2>/dev/null
|
systemctl reset-failed truecloud-mw-restart.service 2>/dev/null
|
||||||
|
|||||||
@@ -0,0 +1,329 @@
|
|||||||
|
"""The *_BLOCK strings in apply.sh are Python source injected into middleware.
|
||||||
|
|
||||||
|
A syntax error in one of them would be appended to a live middlewared module and
|
||||||
|
break the box at boot. They are string literals, so nothing type-checks them --
|
||||||
|
these tests do.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import ast
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import textwrap
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
APPLY_SH = os.path.join(os.path.dirname(__file__), "..", "patch", "apply.sh")
|
||||||
|
|
||||||
|
EXPECTED_BLOCKS = {
|
||||||
|
"B2_BLOCK",
|
||||||
|
"RESTIC_BLOCK",
|
||||||
|
"SNAPSHOT_BLOCK",
|
||||||
|
"CRUD_BLOCK",
|
||||||
|
"SYNC_BLOCK",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def heredoc_source():
|
||||||
|
with open(APPLY_SH, encoding="utf-8") as fh:
|
||||||
|
src = fh.read()
|
||||||
|
m = re.search(r"<< 'PYEOF'\n(.*?)\nPYEOF", src, re.S)
|
||||||
|
assert m, "could not find the PYEOF heredoc in apply.sh"
|
||||||
|
return m.group(1)
|
||||||
|
|
||||||
|
|
||||||
|
def extract_blocks():
|
||||||
|
tree = ast.parse(heredoc_source())
|
||||||
|
blocks = {}
|
||||||
|
for node in ast.walk(tree):
|
||||||
|
if isinstance(node, ast.Assign):
|
||||||
|
for tgt in node.targets:
|
||||||
|
if (
|
||||||
|
isinstance(tgt, ast.Name)
|
||||||
|
and tgt.id.endswith("_BLOCK")
|
||||||
|
and isinstance(node.value, ast.Constant)
|
||||||
|
and isinstance(node.value.value, str)
|
||||||
|
):
|
||||||
|
blocks[tgt.id] = node.value.value
|
||||||
|
return blocks
|
||||||
|
|
||||||
|
|
||||||
|
def _nested_native_detector():
|
||||||
|
"""The REAL native-nested probe, lifted out of apply.sh.
|
||||||
|
|
||||||
|
Extracted rather than reimplemented: a reimplementation would happily pass
|
||||||
|
while the shipped probe stayed broken, which is precisely the bug this guards.
|
||||||
|
"""
|
||||||
|
with open(APPLY_SH, encoding="utf-8") as fh:
|
||||||
|
sh = fh.read()
|
||||||
|
|
||||||
|
m = re.search(
|
||||||
|
r"^(\s*)_drop = str\.maketrans\(.*?\n\s*if 'nofurthernesting' not in "
|
||||||
|
r"stock_src\.translate\(_drop\):\n\s*result\['native_nested'\] = 'yes'",
|
||||||
|
sh, re.S | re.M,
|
||||||
|
)
|
||||||
|
assert m, "could not find the native-nested probe in apply.sh"
|
||||||
|
|
||||||
|
# The block lives inside a double-quoted shell string; undo bash's escaping.
|
||||||
|
body = m.group(0)
|
||||||
|
body = body.replace("\\\\", "\x00").replace('\\"', '"').replace("\x00", "\\")
|
||||||
|
body = textwrap.dedent(body)
|
||||||
|
|
||||||
|
def detect(stock_src):
|
||||||
|
ns = {"stock_src": stock_src, "result": {"native_nested": "no"}, "chr": chr}
|
||||||
|
exec(body, ns) # noqa: S102 - executing our own shipped code, on purpose
|
||||||
|
return ns["result"]["native_nested"]
|
||||||
|
|
||||||
|
return detect
|
||||||
|
|
||||||
|
|
||||||
|
def test_heredoc_itself_compiles():
|
||||||
|
compile(heredoc_source(), "apply.sh:PYEOF", "exec")
|
||||||
|
|
||||||
|
|
||||||
|
def test_all_expected_blocks_present():
|
||||||
|
assert set(extract_blocks()) == EXPECTED_BLOCKS
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("name", sorted(EXPECTED_BLOCKS))
|
||||||
|
def test_injected_block_is_valid_python(name):
|
||||||
|
block = extract_blocks()[name]
|
||||||
|
compile(block, f"apply.sh:{name}", "exec")
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("name", sorted(EXPECTED_BLOCKS))
|
||||||
|
def test_injected_block_carries_the_idempotency_marker(name):
|
||||||
|
# patch_file() truncates each target file at "\n# TRUECLOUD_PATCH" before
|
||||||
|
# re-appending, so every block must start with that marker or repeated runs
|
||||||
|
# would stack duplicate copies into the middleware module.
|
||||||
|
assert extract_blocks()[name].lstrip("\n").startswith("# TRUECLOUD_PATCH")
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("name", ["SNAPSHOT_BLOCK", "CRUD_BLOCK", "SYNC_BLOCK"])
|
||||||
|
def test_nested_blocks_degrade_safely_without_the_module(name):
|
||||||
|
# If _truecloud_nested failed to install, every nested block must no-op.
|
||||||
|
# Critically this includes CRUD_BLOCK: relaxing the guard without the
|
||||||
|
# traversal in place would mean silently-empty backups.
|
||||||
|
block = extract_blocks()[name]
|
||||||
|
assert "_tc_nested = None" in block
|
||||||
|
assert "if _tc_nested is not None:" in block
|
||||||
|
|
||||||
|
|
||||||
|
class TestSnapshotLeak:
|
||||||
|
"""zfs.snapshot.delete is non-recursive and stock calls it with no options.
|
||||||
|
|
||||||
|
A recursive snapshot has one child per descendant dataset (160+ here), so
|
||||||
|
every path that creates one must also sweep the whole tree.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def test_staging_failure_deletes_the_snapshot_tree(self):
|
||||||
|
# On a staging failure, sync.py's `snapshot, local_path = await
|
||||||
|
# create_snapshot(...)` never completes, so its local `snapshot` stays
|
||||||
|
# None and its finally deletes nothing. We must sweep it ourselves.
|
||||||
|
block = extract_blocks()["SNAPSHOT_BLOCK"]
|
||||||
|
assert "except Exception:" in block
|
||||||
|
assert "delete_snapshot_tree" in block
|
||||||
|
assert "raise" in block
|
||||||
|
|
||||||
|
def test_sync_block_cleans_up_on_every_path(self):
|
||||||
|
block = extract_blocks()["SYNC_BLOCK"]
|
||||||
|
assert "finally:" in block
|
||||||
|
assert "cleanup_task" in block
|
||||||
|
|
||||||
|
|
||||||
|
def test_crud_block_is_scoped_to_cloud_backup():
|
||||||
|
# cloudsync has no staging teardown wired in, so its guard must stay.
|
||||||
|
assert '!= "cloud_backup"' in extract_blocks()["CRUD_BLOCK"]
|
||||||
|
|
||||||
|
|
||||||
|
class TestIndependentModules:
|
||||||
|
"""The two modules must retire independently.
|
||||||
|
|
||||||
|
TrueNAS may ship native B2 support long before (or after) it handles nested
|
||||||
|
datasets. A single all-or-nothing kill switch would silently take a
|
||||||
|
still-needed module down with the superseded one.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def _sh(self):
|
||||||
|
with open(APPLY_SH, encoding="utf-8") as fh:
|
||||||
|
return fh.read()
|
||||||
|
|
||||||
|
def test_native_support_is_detected_per_module(self):
|
||||||
|
sh = self._sh()
|
||||||
|
assert "native_b2" in sh
|
||||||
|
assert "native_nested" in sh
|
||||||
|
assert "no further nesting" in sh, "nested native-support probe"
|
||||||
|
|
||||||
|
def test_kill_switch_only_when_both_modules_are_done(self):
|
||||||
|
sh = self._sh()
|
||||||
|
assert '[ "$_providers_needed" = "0" ] && [ "$_nested_needed" = "0" ]' in sh
|
||||||
|
# ...and that is the only place the kill switch is actually set. (Ignore
|
||||||
|
# comment lines, which mention the same path.)
|
||||||
|
code = [ln for ln in sh.splitlines() if not ln.lstrip().startswith("#")]
|
||||||
|
sets = [ln for ln in code if 'touch "$PATCH_DIR/disabled"' in ln]
|
||||||
|
assert len(sets) == 1, f"kill switch set in {len(sets)} places"
|
||||||
|
|
||||||
|
def test_each_module_is_gated_separately(self):
|
||||||
|
src = heredoc_source()
|
||||||
|
assert "if not providers_needed:" in src
|
||||||
|
assert "elif nested_native:" in src
|
||||||
|
|
||||||
|
def test_ui_patch_is_tied_to_the_providers_module(self):
|
||||||
|
# The UI change widens the credential dropdown; it is meaningless once B2
|
||||||
|
# is native, but must NOT be skipped merely because nested is off.
|
||||||
|
sh = self._sh()
|
||||||
|
i = sh.index("--- UI patch ---")
|
||||||
|
assert '[ "$_providers_needed" = "0" ]' in sh[i:i + 400]
|
||||||
|
|
||||||
|
def test_status_reports_an_inactive_module_as_ok(self):
|
||||||
|
# `create_task.py verify` fails if any patches[*].ok is false. An opt-in
|
||||||
|
# module that is switched off (the DEFAULT) must not report FAIL, or a
|
||||||
|
# stock install fails verification out of the box.
|
||||||
|
src = heredoc_source()
|
||||||
|
assert "'ok': (not nested_needed) or nested_ok" in src
|
||||||
|
assert "'ok': (not providers_needed) or bool(b2_ok and restic_ok)" in src
|
||||||
|
assert "'active': nested_needed" in src
|
||||||
|
|
||||||
|
def test_nested_native_probe_matches_the_real_wrapped_source(self):
|
||||||
|
"""Stock splits the guard message across adjacent string literals.
|
||||||
|
|
||||||
|
Python concatenates them at runtime, so the errmsg is contiguous -- but the
|
||||||
|
SOURCE never contains the whole phrase. A raw substring search finds
|
||||||
|
nothing, concludes iX removed the guard, and silently skips this module
|
||||||
|
forever. This is exactly what happened, and only a run against real
|
||||||
|
middlewared caught it.
|
||||||
|
"""
|
||||||
|
detect = _nested_native_detector()
|
||||||
|
|
||||||
|
# Verbatim shape from TrueNAS plugins/cloud/crud.py.
|
||||||
|
stock_wrapped = (
|
||||||
|
' verrors.add(f"{name}.snapshot", '
|
||||||
|
'"This option is only available for datasets that have no further "\n'
|
||||||
|
' "nesting")\n'
|
||||||
|
)
|
||||||
|
assert detect(stock_wrapped) == "no", "guard is present; must NOT report native"
|
||||||
|
|
||||||
|
# Same message on a single line — must also be detected.
|
||||||
|
assert detect('verrors.add(x, "... have no further nesting")\n') == "no"
|
||||||
|
|
||||||
|
# Single-quoted, three-way split — still the guard.
|
||||||
|
assert detect(
|
||||||
|
"verrors.add(x, 'This option is only available for '\n"
|
||||||
|
" 'datasets that have no further '\n"
|
||||||
|
" 'nesting')\n"
|
||||||
|
) == "no"
|
||||||
|
|
||||||
|
# Guard genuinely gone -> native support.
|
||||||
|
assert detect("def _validate(self):\n pass\n") == "yes"
|
||||||
|
|
||||||
|
def test_nested_native_probe_ignores_our_own_block(self):
|
||||||
|
# CRUD_BLOCK quotes the guard message, so scanning the whole file would
|
||||||
|
# find the string in our own patch and never detect native support.
|
||||||
|
sh = self._sh()
|
||||||
|
assert "split('\\n# TRUECLOUD_PATCH', 1)[0]" in sh
|
||||||
|
assert "no further nesting" in extract_blocks()["CRUD_BLOCK"], (
|
||||||
|
"if this ever stops being true, the probe comment is stale"
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_restart_fires_when_any_needed_module_landed(self):
|
||||||
|
# Keying the restart off providers alone would leave a freshly-patched
|
||||||
|
# nested module on disk and never loaded on a native-B2 box.
|
||||||
|
sh = self._sh()
|
||||||
|
i = sh.index("--- deferred restart ---")
|
||||||
|
tail = sh[i:]
|
||||||
|
assert '_backend_ok' in tail
|
||||||
|
assert '"$_b2_ok"' not in tail
|
||||||
|
|
||||||
|
def test_partial_failure_still_schedules_the_restart(self):
|
||||||
|
# If providers fails but nested landed (or vice versa), something new IS
|
||||||
|
# on disk. Collapsing that into "nothing to do" would leave the module
|
||||||
|
# that succeeded permanently unloaded.
|
||||||
|
src = heredoc_source()
|
||||||
|
assert "sys.exit(2 if _landed else 1)" in src
|
||||||
|
assert "_landed = (providers_needed and b2_ok and restic_ok) or (nested_needed and nested_ok)" in src
|
||||||
|
|
||||||
|
sh = self._sh()
|
||||||
|
assert '_rc=$?' in sh
|
||||||
|
assert '[ "$_rc" = "2" ]' in sh
|
||||||
|
|
||||||
|
|
||||||
|
class TestOptIn:
|
||||||
|
"""Nested-snapshot support must be opt-in and must never self-enable."""
|
||||||
|
|
||||||
|
def test_heredoc_gates_on_the_opt_in_flag(self):
|
||||||
|
src = heredoc_source()
|
||||||
|
assert re.search(r"nested_enabled = sys\.argv\[\d+\] == \"1\"", src)
|
||||||
|
assert "if not nested_enabled:" in src
|
||||||
|
|
||||||
|
def test_apply_sh_reads_the_marker_file(self):
|
||||||
|
with open(APPLY_SH, encoding="utf-8") as fh:
|
||||||
|
sh = fh.read()
|
||||||
|
assert 'if [ -f "$PATCH_DIR/nested_snapshots_enabled" ]' in sh
|
||||||
|
assert '"$_NESTED_ENABLED"' in sh
|
||||||
|
|
||||||
|
def test_patching_is_skipped_entirely_when_disabled(self):
|
||||||
|
# The guard-relaxing crud.py patch must be inside the enabled branch.
|
||||||
|
src = heredoc_source()
|
||||||
|
gate = src.index("if not nested_needed:")
|
||||||
|
crud = src.index("patch_file(crud_py, CRUD_BLOCK)")
|
||||||
|
assert gate < crud, "crud.py patch must sit inside the opt-in branch"
|
||||||
|
|
||||||
|
def test_disabling_REVERTS_the_patch_rather_than_merely_skipping_it(self):
|
||||||
|
"""Skipping is not disabling.
|
||||||
|
|
||||||
|
The overlay persists for the whole boot, so a patch applied by an earlier
|
||||||
|
run this boot is still on disk — and middlewared re-imports it on the
|
||||||
|
restart install.sh performs. Without an active revert,
|
||||||
|
`--disable-nested-snapshots` reports "disabled" while the feature keeps
|
||||||
|
running until the next reboot.
|
||||||
|
"""
|
||||||
|
src = heredoc_source()
|
||||||
|
assert "def unpatch_file(" in src
|
||||||
|
assert "def revert_nested(" in src
|
||||||
|
# The revert must run on every not-needed path (opt-out, superseded).
|
||||||
|
gate = src.index("if not nested_needed:")
|
||||||
|
revert = src.index("reverted = revert_nested(")
|
||||||
|
patch = src.index("patch_file(crud_py, CRUD_BLOCK)")
|
||||||
|
assert gate < revert < patch, "revert belongs in the not-needed branch"
|
||||||
|
|
||||||
|
def test_revert_removes_the_module_before_unpatching_files(self):
|
||||||
|
# Every injected block is guarded by `if _tc_nested is not None`, so
|
||||||
|
# deleting the module first means the guard is restored even if a later
|
||||||
|
# unpatch step fails.
|
||||||
|
src = heredoc_source()
|
||||||
|
body = src[src.index("def revert_nested("):src.index("def patch_file(") if
|
||||||
|
src.index("def patch_file(") > src.index("def revert_nested(") else len(src)]
|
||||||
|
body = src[src.index("def revert_nested("):]
|
||||||
|
body = body[:body.index("\n\n\n")] if "\n\n\n" in body else body
|
||||||
|
assert body.index("_truecloud_nested.py") < body.index("crud.py")
|
||||||
|
|
||||||
|
def test_revert_never_touches_the_providers_patch(self):
|
||||||
|
# restic.py also carries a TRUECLOUD_PATCH block, but it belongs to the
|
||||||
|
# providers module. Reverting it would silently break B2 backups.
|
||||||
|
src = heredoc_source()
|
||||||
|
body = src[src.index("def revert_nested("):]
|
||||||
|
body = body[:body.index("return reverted")]
|
||||||
|
# Comments legitimately *mention* restic.py to explain why it is excluded;
|
||||||
|
# what matters is that no code line touches it.
|
||||||
|
code = "\n".join(
|
||||||
|
ln for ln in body.splitlines() if not ln.lstrip().startswith("#")
|
||||||
|
)
|
||||||
|
assert "restic" not in code
|
||||||
|
assert "b2.py" not in code
|
||||||
|
# It must only ever revert these three, plus the module itself.
|
||||||
|
assert "crud.py" in code
|
||||||
|
assert "sync_path" in code
|
||||||
|
assert "snapshot.py" in code
|
||||||
|
|
||||||
|
|
||||||
|
def test_guard_is_relaxed_only_after_traversal_is_installed():
|
||||||
|
# Ordering in apply.sh is a safety property: copy module -> patch snapshot.py
|
||||||
|
# -> patch sync.py -> patch crud.py. crud.py (which unlocks the feature) must
|
||||||
|
# come last, so a partial failure never leaves "guard removed, traversal gone".
|
||||||
|
src = heredoc_source()
|
||||||
|
order = [
|
||||||
|
src.index("shutil.copyfile(nested_src, nested_dst)"),
|
||||||
|
src.index("patch_file(snapshot_py, SNAPSHOT_BLOCK)"),
|
||||||
|
src.index("patch_file(sync_path, SYNC_BLOCK)"),
|
||||||
|
src.index("patch_file(crud_py, CRUD_BLOCK)"),
|
||||||
|
]
|
||||||
|
assert order == sorted(order), "crud.py must be patched last"
|
||||||
@@ -0,0 +1,171 @@
|
|||||||
|
"""Tests for the Angular bundle patch.
|
||||||
|
|
||||||
|
This is the one part of the patch that edits *minified third-party JavaScript* by
|
||||||
|
regex, so it is the easiest place to silently produce a broken bundle: a pattern
|
||||||
|
that matches nothing leaves the dropdown Storj-only, and a pattern that matches
|
||||||
|
sloppily can unbalance the parentheses and take the whole web UI down.
|
||||||
|
|
||||||
|
Nothing checked it until now. The snippets below are verbatim from a real
|
||||||
|
TrueNAS 25.x bundle (chunk-*.js, pre-patch).
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "patch"))
|
||||||
|
|
||||||
|
from patch_ui import MARKER, _PATTERNS, _match_pattern # noqa: E402
|
||||||
|
|
||||||
|
# Verbatim from /usr/share/truenas/webui/chunk-FX2QXNQU.js on TrueNAS 25.10.
|
||||||
|
# Angular emits the binding as a chained ɵɵproperty(...)(...) call, so the
|
||||||
|
# pureFunction call is followed by TWO closing parens: one for pe(...), one for
|
||||||
|
# property(...).
|
||||||
|
REAL_25X = (
|
||||||
|
'c(2,"filterByProviders",pe(115,Rn,i.CloudSyncProviderName.Storj))'
|
||||||
|
'("required",!0),r(3'
|
||||||
|
)
|
||||||
|
|
||||||
|
# TrueNAS 24.x and earlier emitted a literal array.
|
||||||
|
REAL_24X = 'c(2,"filterByProviders",["STORJ_IX"])("required",!0),r(3'
|
||||||
|
|
||||||
|
|
||||||
|
def apply_patch(content):
|
||||||
|
"""Run the same match-and-substitute main() does."""
|
||||||
|
find, replace = _match_pattern(content)
|
||||||
|
assert find is not None, "no pattern matched"
|
||||||
|
patched, count = find.subn(replace, content)
|
||||||
|
return patched, count
|
||||||
|
|
||||||
|
|
||||||
|
def paren_delta(s):
|
||||||
|
"""Net paren balance. The snippets are fragments of a minified file, so they
|
||||||
|
are not balanced on their own -- what must hold is that patching does not
|
||||||
|
CHANGE the balance. Consuming one paren too many is a syntax error in the
|
||||||
|
bundle, and the whole TrueNAS web UI goes blank."""
|
||||||
|
return s.count("(") - s.count(")")
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("source", [REAL_25X, REAL_24X], ids=["25.x", "24.x"])
|
||||||
|
class TestAgainstRealBundles:
|
||||||
|
def test_matches_exactly_once(self, source):
|
||||||
|
# main() refuses to write unless count == 1 — more than one match would
|
||||||
|
# mean the pattern is too loose to trust against a minified bundle.
|
||||||
|
_patched, count = apply_patch(source)
|
||||||
|
assert count == 1
|
||||||
|
|
||||||
|
def test_result_contains_all_three_providers(self, source):
|
||||||
|
patched, _ = apply_patch(source)
|
||||||
|
assert MARKER in patched
|
||||||
|
assert '"filterByProviders",["STORJ_IX","S3","B2"]' in patched
|
||||||
|
|
||||||
|
def test_patch_does_not_change_paren_balance(self, source):
|
||||||
|
# Consuming one paren too many (or too few) is a syntax error in the
|
||||||
|
# bundle and the entire TrueNAS web UI goes blank. This is the invariant
|
||||||
|
# the 25.x pattern has to get right: it eats `pe(...)` which sits inside
|
||||||
|
# a chained property(...)(...) call.
|
||||||
|
patched, _ = apply_patch(source)
|
||||||
|
assert paren_delta(patched) == paren_delta(source)
|
||||||
|
|
||||||
|
def test_surrounding_code_is_untouched(self, source):
|
||||||
|
patched, _ = apply_patch(source)
|
||||||
|
assert patched.startswith("c(2,")
|
||||||
|
assert patched.endswith('("required",!0),r(3')
|
||||||
|
|
||||||
|
def test_patch_is_idempotent(self, source):
|
||||||
|
# apply.sh re-runs every boot; MARKER short-circuits an already-patched
|
||||||
|
# file, but the pattern must also not match its own output.
|
||||||
|
patched, _ = apply_patch(source)
|
||||||
|
find, _replace = _match_pattern(patched)
|
||||||
|
if find is not None:
|
||||||
|
# Only the 24.x literal-array pattern may still "match" — and only if
|
||||||
|
# it would produce the same text. Anything else means double-patching.
|
||||||
|
again, _ = apply_patch(patched)
|
||||||
|
assert again == patched, "re-patching must be a no-op"
|
||||||
|
|
||||||
|
|
||||||
|
def test_storj_only_bundle_is_recognised():
|
||||||
|
assert _match_pattern(REAL_25X)[0] is not None
|
||||||
|
|
||||||
|
|
||||||
|
def test_unrelated_javascript_is_never_touched():
|
||||||
|
# A pattern loose enough to hit unrelated code would corrupt the bundle.
|
||||||
|
for noise in (
|
||||||
|
'c(2,"filterByProviders",pe(115,Rn,i.SomethingElse.Storj))',
|
||||||
|
'c(2,"otherBinding",pe(115,Rn,i.CloudSyncProviderName.Storj))',
|
||||||
|
'"filterByProviders"',
|
||||||
|
):
|
||||||
|
find, _ = _match_pattern(noise)
|
||||||
|
assert find is None, f"pattern must not match: {noise}"
|
||||||
|
|
||||||
|
|
||||||
|
def test_every_pattern_is_anchored_to_filterbyproviders():
|
||||||
|
# Guards against a future pattern broad enough to rewrite arbitrary JS.
|
||||||
|
for find, _replace in _PATTERNS:
|
||||||
|
assert "filterByProviders" in find.pattern
|
||||||
|
|
||||||
|
|
||||||
|
def test_patterns_compile_and_replacements_reference_group_one():
|
||||||
|
for find, replace in _PATTERNS:
|
||||||
|
assert isinstance(find, re.Pattern)
|
||||||
|
assert r"\1" in replace, "replacement must preserve the binding name"
|
||||||
|
|
||||||
|
|
||||||
|
class TestCorruptionGuard:
|
||||||
|
"""A bad pattern must never reach the bundle.
|
||||||
|
|
||||||
|
This is not hypothetical. Commit 47cdf72 shipped a pattern that consumed one
|
||||||
|
closing paren and emitted one, netting an extra `)`:
|
||||||
|
|
||||||
|
c(2,"filterByProviders",["STORJ_IX","S3","B2"]))("required",!0)
|
||||||
|
^^ syntax error
|
||||||
|
|
||||||
|
The web UI went blank. And because MARKER was then present in the file, every
|
||||||
|
subsequent run reported "already patched" and skipped — so the patch could not
|
||||||
|
heal itself, and the bundle had to be hand-restored from the backup.
|
||||||
|
"""
|
||||||
|
|
||||||
|
# Verbatim from 47cdf72.
|
||||||
|
BROKEN = (
|
||||||
|
re.compile(r'("filterByProviders",)\w+\(\d+,\w+,\w+\.CloudSyncProviderName\.Storj\)'),
|
||||||
|
r'\1["STORJ_IX","S3","B2"])',
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_the_regression_that_blanked_the_ui_is_detectable(self):
|
||||||
|
find, replace = self.BROKEN
|
||||||
|
patched, count = find.subn(replace, REAL_25X)
|
||||||
|
assert count == 1, "it did match — that is why it got written"
|
||||||
|
assert paren_delta(patched) != paren_delta(REAL_25X), (
|
||||||
|
"the paren balance changes; this is the signal main() now refuses on"
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_main_refuses_to_write_an_unbalanced_bundle(self, monkeypatch, tmp_path, capsys):
|
||||||
|
import patch_ui
|
||||||
|
|
||||||
|
bundle = tmp_path / "chunk-TEST.js"
|
||||||
|
bundle.write_text(REAL_25X, encoding="utf-8")
|
||||||
|
|
||||||
|
monkeypatch.setattr(patch_ui, "WEBUI_CANDIDATES", [str(tmp_path)])
|
||||||
|
monkeypatch.setattr(patch_ui, "_PATTERNS", [self.BROKEN])
|
||||||
|
|
||||||
|
patch_ui.main()
|
||||||
|
|
||||||
|
out = capsys.readouterr().out
|
||||||
|
assert "refusing to write" in out
|
||||||
|
# The bundle must be byte-for-byte untouched — a broken UI is far worse
|
||||||
|
# than an unpatched one.
|
||||||
|
assert bundle.read_text(encoding="utf-8") == REAL_25X
|
||||||
|
|
||||||
|
def test_a_good_pattern_still_writes(self, monkeypatch, tmp_path):
|
||||||
|
import patch_ui
|
||||||
|
|
||||||
|
bundle = tmp_path / "chunk-TEST.js"
|
||||||
|
bundle.write_text(REAL_25X, encoding="utf-8")
|
||||||
|
monkeypatch.setattr(patch_ui, "WEBUI_CANDIDATES", [str(tmp_path)])
|
||||||
|
|
||||||
|
patch_ui.main()
|
||||||
|
|
||||||
|
assert MARKER in bundle.read_text(encoding="utf-8")
|
||||||
|
assert (tmp_path / "chunk-TEST.js.pre-truecloud-patch").exists()
|
||||||
@@ -0,0 +1,122 @@
|
|||||||
|
"""Tests for the release automation.
|
||||||
|
|
||||||
|
The release workflow refuses to publish unless these hold, so a bad tag fails
|
||||||
|
loudly in CI instead of shipping a release whose notes are empty, wrong, or whose
|
||||||
|
scripts announce a different version than the tag.
|
||||||
|
|
||||||
|
That last one is not hypothetical: VERSION= drifted to three different values
|
||||||
|
across install.sh / uninstall.sh / recover.sh / apply.sh and nothing noticed.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "tools"))
|
||||||
|
|
||||||
|
from release_notes import ( # noqa: E402
|
||||||
|
changelog_versions,
|
||||||
|
check,
|
||||||
|
extract_notes,
|
||||||
|
normalise,
|
||||||
|
script_versions,
|
||||||
|
)
|
||||||
|
|
||||||
|
REPO = os.path.join(os.path.dirname(__file__), "..")
|
||||||
|
|
||||||
|
SAMPLE = """\
|
||||||
|
# Changelog
|
||||||
|
|
||||||
|
## v0.3.0 — 2026-07-13
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- the new thing
|
||||||
|
|
||||||
|
## v0.2.1 — 2026-07-09
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- the old thing
|
||||||
|
|
||||||
|
## v0.2.0 — 2026-07-08
|
||||||
|
|
||||||
|
- first
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
class TestExtractNotes:
|
||||||
|
def test_returns_only_that_versions_body(self):
|
||||||
|
body = extract_notes(SAMPLE, "v0.3.0")
|
||||||
|
assert "the new thing" in body
|
||||||
|
assert "the old thing" not in body
|
||||||
|
# The version heading itself is dropped (GitHub renders its own title),
|
||||||
|
# but sub-headings like "### Added" must survive.
|
||||||
|
assert not body.startswith("## v")
|
||||||
|
assert body.startswith("### Added")
|
||||||
|
|
||||||
|
def test_stops_at_the_next_version_heading(self):
|
||||||
|
body = extract_notes(SAMPLE, "v0.2.1")
|
||||||
|
assert "the old thing" in body
|
||||||
|
assert "first" not in body
|
||||||
|
|
||||||
|
def test_last_section_runs_to_end_of_file(self):
|
||||||
|
assert "first" in extract_notes(SAMPLE, "v0.2.0")
|
||||||
|
|
||||||
|
def test_accepts_the_tag_with_or_without_the_v(self):
|
||||||
|
assert extract_notes(SAMPLE, "0.3.0") == extract_notes(SAMPLE, "v0.3.0")
|
||||||
|
|
||||||
|
def test_unknown_version_raises_rather_than_returning_empty(self):
|
||||||
|
# An empty release body is worse than a failed release.
|
||||||
|
with pytest.raises(KeyError, match="no section"):
|
||||||
|
extract_notes(SAMPLE, "v9.9.9")
|
||||||
|
|
||||||
|
|
||||||
|
class TestChangelogVersions:
|
||||||
|
def test_lists_versions_newest_first(self):
|
||||||
|
assert changelog_versions(SAMPLE) == ["0.3.0", "0.2.1", "0.2.0"]
|
||||||
|
|
||||||
|
|
||||||
|
class TestAgainstTheRealRepo:
|
||||||
|
"""These run against the actual files, so drift breaks the build."""
|
||||||
|
|
||||||
|
def test_every_script_declares_a_version(self):
|
||||||
|
from release_notes import VERSIONED_FILES
|
||||||
|
|
||||||
|
found = script_versions(REPO)
|
||||||
|
missing = [f for f in VERSIONED_FILES if f not in found]
|
||||||
|
assert not missing, f"no VERSION= in: {missing}"
|
||||||
|
|
||||||
|
def test_all_scripts_agree_on_the_version(self):
|
||||||
|
versions = {normalise(v) for v in script_versions(REPO).values()}
|
||||||
|
assert len(versions) == 1, f"scripts disagree on version: {sorted(versions)}"
|
||||||
|
|
||||||
|
def test_the_current_version_has_a_changelog_section(self):
|
||||||
|
version = next(iter({normalise(v) for v in script_versions(REPO).values()}))
|
||||||
|
with open(os.path.join(REPO, "CHANGELOG.md"), encoding="utf-8") as fh:
|
||||||
|
body = extract_notes(fh.read(), version)
|
||||||
|
assert body, f"CHANGELOG.md has no content for v{version}"
|
||||||
|
|
||||||
|
def test_the_current_version_is_the_newest_changelog_entry(self):
|
||||||
|
version = next(iter({normalise(v) for v in script_versions(REPO).values()}))
|
||||||
|
with open(os.path.join(REPO, "CHANGELOG.md"), encoding="utf-8") as fh:
|
||||||
|
newest = changelog_versions(fh.read())[0]
|
||||||
|
assert newest == version, (
|
||||||
|
f"scripts say v{version} but the newest CHANGELOG entry is v{newest}"
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_check_passes_for_the_current_version(self):
|
||||||
|
version = next(iter({normalise(v) for v in script_versions(REPO).values()}))
|
||||||
|
assert check(version, REPO) == []
|
||||||
|
|
||||||
|
|
||||||
|
class TestCheckCatchesMistakes:
|
||||||
|
def test_reports_a_tag_that_no_script_matches(self):
|
||||||
|
problems = check("v9.9.9", REPO)
|
||||||
|
assert problems
|
||||||
|
assert any("declares VERSION" in p for p in problems)
|
||||||
|
|
||||||
|
def test_reports_a_missing_changelog_section(self):
|
||||||
|
problems = check("v9.9.9", REPO)
|
||||||
|
assert any("no section" in p for p in problems)
|
||||||
@@ -0,0 +1,605 @@
|
|||||||
|
"""Tests for nested-dataset snapshot staging.
|
||||||
|
|
||||||
|
Two rules are under test above all else:
|
||||||
|
|
||||||
|
1. A tree that cannot be staged completely must fail LOUDLY. A silently
|
||||||
|
incomplete backup is the exact failure that stock TrueNAS's "no further
|
||||||
|
nesting" guard exists to prevent.
|
||||||
|
|
||||||
|
2. Every snapshot we cause to exist must be cleaned up. ``zfs.snapshot.delete``
|
||||||
|
is non-recursive by default and stock calls it with no options, so a
|
||||||
|
recursive snapshot would otherwise orphan one snapshot per descendant dataset
|
||||||
|
on EVERY run.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "patch"))
|
||||||
|
|
||||||
|
from truecloud_nested import ( # noqa: E402
|
||||||
|
StagingError,
|
||||||
|
apply_plan,
|
||||||
|
cleanup_all,
|
||||||
|
cleanup_task,
|
||||||
|
current_mounts_under,
|
||||||
|
delete_snapshot_tree,
|
||||||
|
plan_staging,
|
||||||
|
sidecar_for,
|
||||||
|
snapshot_tree_names,
|
||||||
|
staging_root_for,
|
||||||
|
teardown,
|
||||||
|
verify_staged,
|
||||||
|
)
|
||||||
|
|
||||||
|
SNAP = "cloud_backup-5-20260712030000"
|
||||||
|
ROOT = "/run/truecloud-nested/cloud_backup-5"
|
||||||
|
|
||||||
|
|
||||||
|
def ds(name, mountpoint, mounted="yes"):
|
||||||
|
return {
|
||||||
|
"name": name,
|
||||||
|
"properties": {
|
||||||
|
"mountpoint": {"value": mountpoint},
|
||||||
|
"mounted": {"value": mounted},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
# Mirrors the real layout: apps are datasets, several with their own children.
|
||||||
|
DATASETS = [
|
||||||
|
ds("Tap", "/mnt/Tap"),
|
||||||
|
ds("Tap/apps", "/mnt/Tap/apps"),
|
||||||
|
ds("Tap/apps/lidarr", "/mnt/Tap/apps/lidarr"),
|
||||||
|
ds("Tap/apps/lidarr/config", "/mnt/Tap/apps/lidarr/config"),
|
||||||
|
ds("Tap/apps/immich", "/mnt/Tap/apps/immich"),
|
||||||
|
ds("Tap/apps/immich/pgdata", "/mnt/Tap/apps/immich/pgdata"),
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def yes(_path):
|
||||||
|
return "ok"
|
||||||
|
|
||||||
|
|
||||||
|
def plan(datasets=DATASETS, base_dataset="Tap", base_mp="/mnt/Tap",
|
||||||
|
path="/mnt/Tap", probe=yes):
|
||||||
|
return plan_staging(base_dataset, base_mp, path, SNAP, datasets, ROOT, probe=probe)
|
||||||
|
|
||||||
|
|
||||||
|
class TestPlanStaging:
|
||||||
|
def test_stages_every_descendant_dataset(self):
|
||||||
|
mounts, skipped = plan()
|
||||||
|
assert skipped == []
|
||||||
|
assert len(mounts) == 6 # root + 5 descendants
|
||||||
|
assert mounts[0] == (f"/mnt/Tap/.zfs/snapshot/{SNAP}", ROOT)
|
||||||
|
|
||||||
|
by_target = {t: s for s, t in mounts}
|
||||||
|
assert by_target[f"{ROOT}/apps"] == f"/mnt/Tap/apps/.zfs/snapshot/{SNAP}"
|
||||||
|
assert by_target[f"{ROOT}/apps/immich/pgdata"] == (
|
||||||
|
f"/mnt/Tap/apps/immich/pgdata/.zfs/snapshot/{SNAP}"
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_parents_are_mounted_before_children(self):
|
||||||
|
# A child's mountpoint dir only exists inside its parent's snapshot, so
|
||||||
|
# mounting a child first would fail.
|
||||||
|
mounts, _ = plan()
|
||||||
|
seen = set()
|
||||||
|
for _src, target in mounts:
|
||||||
|
if target != ROOT:
|
||||||
|
assert os.path.dirname(target) in seen
|
||||||
|
seen.add(target)
|
||||||
|
|
||||||
|
def test_backup_path_below_dataset_root(self):
|
||||||
|
mounts, _ = plan(base_dataset="Tap/apps", base_mp="/mnt/Tap/apps",
|
||||||
|
path="/mnt/Tap/apps")
|
||||||
|
assert mounts[0] == (f"/mnt/Tap/apps/.zfs/snapshot/{SNAP}", ROOT)
|
||||||
|
targets = [t for _s, t in mounts]
|
||||||
|
assert f"{ROOT}/lidarr" in targets
|
||||||
|
assert f"{ROOT}/apps/lidarr" not in targets
|
||||||
|
|
||||||
|
def test_base_dataset_is_not_a_descendant_of_itself(self):
|
||||||
|
mounts, _ = plan(datasets=[ds("Tap", "/mnt/Tap")])
|
||||||
|
assert len(mounts) == 1
|
||||||
|
|
||||||
|
|
||||||
|
class TestScoping:
|
||||||
|
def test_unrelated_datasets_are_ignored_silently(self):
|
||||||
|
# Regression: scoping by mountpoint first dragged in every
|
||||||
|
# mountpoint-less dataset on the box (all of Tank/.system/*), burying the
|
||||||
|
# warnings that actually matter.
|
||||||
|
noisy = DATASETS + [
|
||||||
|
ds("Tank/.system", "none"),
|
||||||
|
ds("Tank/.system/cores", "legacy"),
|
||||||
|
ds("Tank/backups", "/mnt/Tank/backups"),
|
||||||
|
]
|
||||||
|
mounts, skipped = plan(datasets=noisy)
|
||||||
|
assert len(mounts) == 6
|
||||||
|
assert skipped == [], "datasets outside the base dataset must not be reported"
|
||||||
|
|
||||||
|
def test_in_scope_dataset_without_mountpoint_is_reported(self):
|
||||||
|
datasets = DATASETS + [ds("Tap/apps/weird", "none")]
|
||||||
|
_mounts, skipped = plan(datasets=datasets)
|
||||||
|
assert ("Tap/apps/weird", "mountpoint is none") in skipped
|
||||||
|
|
||||||
|
def test_unmounted_dataset_is_skipped_but_never_silently(self):
|
||||||
|
datasets = DATASETS + [ds("Tap/apps/vault", "/mnt/Tap/apps/vault", mounted="no")]
|
||||||
|
mounts, skipped = plan(datasets=datasets)
|
||||||
|
assert f"{ROOT}/apps/vault" not in [t for _s, t in mounts]
|
||||||
|
assert ("Tap/apps/vault", "dataset is not mounted (locked/encrypted?)") in skipped
|
||||||
|
|
||||||
|
def test_descendant_mounted_outside_the_path_is_not_an_omission(self):
|
||||||
|
datasets = DATASETS + [ds("Tap/elsewhere", "/mnt/other")]
|
||||||
|
mounts, skipped = plan(datasets=datasets)
|
||||||
|
assert len(mounts) == 6
|
||||||
|
assert skipped == []
|
||||||
|
|
||||||
|
|
||||||
|
class TestSilentOmissionGuard:
|
||||||
|
"""The whole point of the feature. These are the tests that matter."""
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _missing_pgdata(path):
|
||||||
|
return "missing" if "/mnt/Tap/apps/immich/pgdata/" in path else "ok"
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _denied_pgdata(path):
|
||||||
|
if "/mnt/Tap/apps/immich/pgdata/" in path:
|
||||||
|
return "cannot be read (Permission denied)"
|
||||||
|
return "ok"
|
||||||
|
|
||||||
|
def test_missing_snapshot_on_descendant_raises(self):
|
||||||
|
with pytest.raises(StagingError, match="incomplete tree"):
|
||||||
|
plan(probe=self._missing_pgdata)
|
||||||
|
|
||||||
|
def test_error_names_the_offending_dataset(self):
|
||||||
|
with pytest.raises(StagingError, match="Tap/apps/immich/pgdata"):
|
||||||
|
plan(probe=self._missing_pgdata)
|
||||||
|
|
||||||
|
def test_missing_and_unreadable_are_reported_differently(self):
|
||||||
|
# os.path.isdir() collapses both into False, which would report a
|
||||||
|
# permission problem as "has no snapshot" and send you hunting for a
|
||||||
|
# snapshot that is sitting right there. Both abort -- but say which.
|
||||||
|
with pytest.raises(StagingError, match="has no snapshot"):
|
||||||
|
plan(probe=self._missing_pgdata)
|
||||||
|
with pytest.raises(StagingError, match="Permission denied"):
|
||||||
|
plan(probe=self._denied_pgdata)
|
||||||
|
|
||||||
|
|
||||||
|
class TestSnapshotTreeNames:
|
||||||
|
"""zfs.snapshot.delete is non-recursive; we must sweep children ourselves."""
|
||||||
|
|
||||||
|
ALL = [
|
||||||
|
"Tap@cloud_backup-5-20260712030000",
|
||||||
|
"Tap/apps@cloud_backup-5-20260712030000",
|
||||||
|
"Tap/apps/lidarr/config@cloud_backup-5-20260712030000",
|
||||||
|
"Tap@auto-2026-07-12_03-00", # unrelated periodic snapshot
|
||||||
|
"Tap/apps@cloud_backup-9-20260712030000", # another task
|
||||||
|
"Tank/backups@cloud_backup-5-20260712030000", # different pool
|
||||||
|
]
|
||||||
|
|
||||||
|
def test_returns_parent_and_all_children(self):
|
||||||
|
got = snapshot_tree_names("Tap@cloud_backup-5-20260712030000", self.ALL)
|
||||||
|
assert set(got) == {
|
||||||
|
"Tap@cloud_backup-5-20260712030000",
|
||||||
|
"Tap/apps@cloud_backup-5-20260712030000",
|
||||||
|
"Tap/apps/lidarr/config@cloud_backup-5-20260712030000",
|
||||||
|
}
|
||||||
|
|
||||||
|
def test_never_touches_periodic_or_other_tasks_or_other_pools(self):
|
||||||
|
got = snapshot_tree_names("Tap@cloud_backup-5-20260712030000", self.ALL)
|
||||||
|
assert "Tap@auto-2026-07-12_03-00" not in got
|
||||||
|
assert "Tap/apps@cloud_backup-9-20260712030000" not in got
|
||||||
|
assert "Tank/backups@cloud_backup-5-20260712030000" not in got
|
||||||
|
|
||||||
|
def test_malformed_snapshot_name_yields_nothing(self):
|
||||||
|
assert snapshot_tree_names("Tap", self.ALL) == []
|
||||||
|
|
||||||
|
|
||||||
|
class FakeMiddleware:
|
||||||
|
def __init__(self, snapshots=None):
|
||||||
|
self.snapshots = list(snapshots or [])
|
||||||
|
self.calls = []
|
||||||
|
self.logger = None
|
||||||
|
|
||||||
|
async def call(self, method, *args):
|
||||||
|
self.calls.append((method, args))
|
||||||
|
if method == "zfs.snapshot.query":
|
||||||
|
return [{"name": n} for n in self.snapshots]
|
||||||
|
if method == "zfs.snapshot.delete":
|
||||||
|
name = args[0]
|
||||||
|
opts = args[1] if len(args) > 1 else {}
|
||||||
|
if name not in self.snapshots:
|
||||||
|
raise RuntimeError("does not exist")
|
||||||
|
if opts.get("recursive"):
|
||||||
|
# Real `zfs destroy -r` takes the parent and every child snapshot.
|
||||||
|
for n in snapshot_tree_names(name, list(self.snapshots)):
|
||||||
|
self.snapshots.remove(n)
|
||||||
|
else:
|
||||||
|
self.snapshots.remove(name)
|
||||||
|
return True
|
||||||
|
raise AssertionError(f"unexpected call {method}")
|
||||||
|
|
||||||
|
async def run_in_thread(self, fn, *args):
|
||||||
|
return fn(*args)
|
||||||
|
|
||||||
|
|
||||||
|
class TestDeleteSnapshotTree:
|
||||||
|
def test_deletes_parent_and_every_child(self):
|
||||||
|
mw = FakeMiddleware([
|
||||||
|
"Tap@snap", "Tap/apps@snap", "Tap/apps/lidarr@snap", "Tap@keepme",
|
||||||
|
])
|
||||||
|
asyncio.run(delete_snapshot_tree(mw, "Tap@snap"))
|
||||||
|
assert mw.snapshots == ["Tap@keepme"]
|
||||||
|
|
||||||
|
def test_is_idempotent_when_stock_already_removed_the_parent(self):
|
||||||
|
# Stock's finally can win the race once our mounts are released.
|
||||||
|
mw = FakeMiddleware(["Tap/apps@snap", "Tap/apps/lidarr@snap"])
|
||||||
|
asyncio.run(delete_snapshot_tree(mw, "Tap@snap"))
|
||||||
|
assert mw.snapshots == []
|
||||||
|
|
||||||
|
def test_uses_a_single_recursive_delete_not_252_individual_ones(self):
|
||||||
|
# 252 sequential deletes are slow AND not atomic: a run killed part-way
|
||||||
|
# through leaves exactly the orphans this function exists to prevent.
|
||||||
|
mw = FakeMiddleware(["Tap@snap", "Tap/apps@snap", "Tap/apps/lidarr@snap"])
|
||||||
|
asyncio.run(delete_snapshot_tree(mw, "Tap@snap"))
|
||||||
|
assert mw.snapshots == []
|
||||||
|
deletes = [a for m, a in mw.calls if m == "zfs.snapshot.delete"]
|
||||||
|
assert len(deletes) == 1, "should be ONE recursive call, not one per snapshot"
|
||||||
|
assert deletes[0][1] == {"recursive": True}
|
||||||
|
assert not [m for m, _a in mw.calls if m == "zfs.snapshot.query"], (
|
||||||
|
"no enumeration needed on the fast path"
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_survives_recursive_and_query_failure_by_deleting_the_parent(self):
|
||||||
|
class Broken(FakeMiddleware):
|
||||||
|
async def call(self, method, *args):
|
||||||
|
if method == "zfs.snapshot.query":
|
||||||
|
raise RuntimeError("boom")
|
||||||
|
if method == "zfs.snapshot.delete" and len(args) > 1:
|
||||||
|
raise RuntimeError("recursive delete unavailable")
|
||||||
|
return await super().call(method, *args)
|
||||||
|
|
||||||
|
mw = Broken(["Tap@snap"])
|
||||||
|
asyncio.run(delete_snapshot_tree(mw, "Tap@snap"))
|
||||||
|
assert mw.snapshots == []
|
||||||
|
|
||||||
|
def test_leaves_unrelated_snapshots_alone_when_the_tree_is_gone(self):
|
||||||
|
mw = FakeMiddleware(["Tap@unrelated"])
|
||||||
|
asyncio.run(delete_snapshot_tree(mw, "Tap@snap"))
|
||||||
|
assert mw.snapshots == ["Tap@unrelated"]
|
||||||
|
|
||||||
|
|
||||||
|
class TestStageNestedOrdering:
|
||||||
|
def test_sidecar_is_written_before_anything_is_mounted(self, tmp_path, monkeypatch):
|
||||||
|
# middlewared can die at any moment. If the snapshot were recorded only
|
||||||
|
# after apply_plan, a crash in that window would orphan a 160-snapshot
|
||||||
|
# tree -- the precise failure the sidecar exists to prevent.
|
||||||
|
import truecloud_nested as tn
|
||||||
|
|
||||||
|
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
|
||||||
|
order = []
|
||||||
|
|
||||||
|
class Recorder(FakeMiddleware):
|
||||||
|
async def run_in_thread(self, fn, *args):
|
||||||
|
order.append(fn.__name__)
|
||||||
|
if fn.__name__ == "plan_staging":
|
||||||
|
return ([("/src", str(tmp_path / "cloud_backup-5"))], [])
|
||||||
|
if fn.__name__ in ("apply_plan", "verify_staged", "teardown"):
|
||||||
|
return [] if fn.__name__ == "teardown" else True
|
||||||
|
return fn(*args)
|
||||||
|
|
||||||
|
asyncio.run(tn.stage_nested(
|
||||||
|
Recorder(), "/mnt/Tap", "Tap@snap", "Tap", "/mnt/Tap",
|
||||||
|
"cloud_backup-5", DATASETS,
|
||||||
|
))
|
||||||
|
|
||||||
|
assert order.index("_write_sidecar") < order.index("apply_plan")
|
||||||
|
|
||||||
|
def test_reclaims_the_snapshot_tree_left_by_a_crashed_run(self, tmp_path,
|
||||||
|
monkeypatch):
|
||||||
|
# teardown() reclaims the crashed run's MOUNTS, but nothing else would
|
||||||
|
# ever reclaim its SNAPSHOTS -- and we are about to overwrite the only
|
||||||
|
# record of them. One crash would orphan 160+ snapshots permanently.
|
||||||
|
import truecloud_nested as tn
|
||||||
|
|
||||||
|
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
|
||||||
|
root = tn.staging_root_for("cloud_backup-5")
|
||||||
|
os.makedirs(os.path.dirname(root), exist_ok=True)
|
||||||
|
with open(sidecar_for(root), "w", encoding="utf-8") as fh:
|
||||||
|
fh.write("Tap@old-crashed-run")
|
||||||
|
|
||||||
|
mw = FakeMiddleware(["Tap@old-crashed-run", "Tap/apps@old-crashed-run"])
|
||||||
|
|
||||||
|
class Stub(FakeMiddleware):
|
||||||
|
def __init__(self, inner):
|
||||||
|
super().__init__()
|
||||||
|
self.inner = inner
|
||||||
|
|
||||||
|
async def call(self, method, *args):
|
||||||
|
return await self.inner.call(method, *args)
|
||||||
|
|
||||||
|
async def run_in_thread(self, fn, *args):
|
||||||
|
if fn.__name__ == "plan_staging":
|
||||||
|
return ([("/src", root)], [])
|
||||||
|
if fn.__name__ == "teardown":
|
||||||
|
return []
|
||||||
|
if fn.__name__ in ("apply_plan", "verify_staged"):
|
||||||
|
return True
|
||||||
|
return fn(*args)
|
||||||
|
|
||||||
|
asyncio.run(tn.stage_nested(
|
||||||
|
Stub(mw), "/mnt/Tap", "Tap@new", "Tap", "/mnt/Tap",
|
||||||
|
"cloud_backup-5", DATASETS,
|
||||||
|
))
|
||||||
|
|
||||||
|
assert mw.snapshots == [], "the crashed run's snapshot tree must be reclaimed"
|
||||||
|
|
||||||
|
def test_sidecar_is_removed_when_staging_fails(self, tmp_path, monkeypatch):
|
||||||
|
import truecloud_nested as tn
|
||||||
|
|
||||||
|
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
|
||||||
|
root = tn.staging_root_for("cloud_backup-5")
|
||||||
|
|
||||||
|
class Failing(FakeMiddleware):
|
||||||
|
async def run_in_thread(self, fn, *args):
|
||||||
|
if fn.__name__ == "plan_staging":
|
||||||
|
raise StagingError("boom")
|
||||||
|
return fn(*args)
|
||||||
|
|
||||||
|
with pytest.raises(StagingError):
|
||||||
|
asyncio.run(tn.stage_nested(
|
||||||
|
Failing(), "/mnt/Tap", "Tap@snap", "Tap", "/mnt/Tap",
|
||||||
|
"cloud_backup-5", DATASETS,
|
||||||
|
))
|
||||||
|
|
||||||
|
assert not os.path.exists(sidecar_for(root))
|
||||||
|
|
||||||
|
|
||||||
|
class TestCleanupTask:
|
||||||
|
def test_recovers_snapshot_from_sidecar_after_middlewared_restart(self, tmp_path,
|
||||||
|
monkeypatch):
|
||||||
|
# The sidecar is the ONLY record of the pinned snapshot, precisely so a
|
||||||
|
# middlewared restart cannot orphan the tree.
|
||||||
|
import truecloud_nested as tn
|
||||||
|
|
||||||
|
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
|
||||||
|
root = tn.staging_root_for("cloud_backup-5", base=str(tmp_path))
|
||||||
|
os.makedirs(root, exist_ok=True)
|
||||||
|
with open(sidecar_for(root), "w", encoding="utf-8") as fh:
|
||||||
|
fh.write("Tap@snap")
|
||||||
|
|
||||||
|
mw = FakeMiddleware(["Tap@snap", "Tap/apps@snap"])
|
||||||
|
monkeypatch.setattr(tn, "teardown", lambda *_a, **_k: [])
|
||||||
|
|
||||||
|
asyncio.run(cleanup_task(mw, "cloud_backup-5"))
|
||||||
|
|
||||||
|
assert mw.snapshots == []
|
||||||
|
assert not os.path.exists(sidecar_for(root))
|
||||||
|
|
||||||
|
def test_is_a_noop_when_never_staged(self, tmp_path, monkeypatch):
|
||||||
|
import truecloud_nested as tn
|
||||||
|
|
||||||
|
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path / "nope"))
|
||||||
|
mw = FakeMiddleware(["Tap@snap"])
|
||||||
|
asyncio.run(cleanup_task(mw, "cloud_backup-5"))
|
||||||
|
assert mw.calls == []
|
||||||
|
assert mw.snapshots == ["Tap@snap"]
|
||||||
|
|
||||||
|
|
||||||
|
class TestVerifyStaged:
|
||||||
|
"""Anti-regression guard: proves the staged tree is real before we back it up."""
|
||||||
|
|
||||||
|
def test_passes_when_every_target_is_mounted_and_root_non_empty(self):
|
||||||
|
mounts = [("/src", ROOT), ("/src/a", f"{ROOT}/a")]
|
||||||
|
assert verify_staged(mounts, ismount=lambda p: True, listdir=lambda p: ["apps"])
|
||||||
|
|
||||||
|
def test_raises_when_a_target_is_not_actually_mounted(self):
|
||||||
|
# This is the case that would produce a silently-empty backup.
|
||||||
|
mounts = [("/src", ROOT), ("/src/a", f"{ROOT}/a")]
|
||||||
|
with pytest.raises(StagingError, match="not a mountpoint"):
|
||||||
|
verify_staged(mounts, ismount=lambda p: p == ROOT, listdir=lambda p: ["apps"])
|
||||||
|
|
||||||
|
def test_raises_when_staging_root_is_empty(self):
|
||||||
|
with pytest.raises(StagingError, match="empty"):
|
||||||
|
verify_staged([("/src", ROOT)], ismount=lambda p: True, listdir=lambda p: [])
|
||||||
|
|
||||||
|
def test_raises_on_empty_plan(self):
|
||||||
|
with pytest.raises(StagingError):
|
||||||
|
verify_staged([])
|
||||||
|
|
||||||
|
|
||||||
|
class FakeRunner:
|
||||||
|
def __init__(self, fail_on=None):
|
||||||
|
self.fail_on = fail_on
|
||||||
|
self.calls = []
|
||||||
|
|
||||||
|
def __call__(self, cmd):
|
||||||
|
self.calls.append(cmd)
|
||||||
|
|
||||||
|
class R:
|
||||||
|
returncode = 0
|
||||||
|
stderr = ""
|
||||||
|
|
||||||
|
if self.fail_on and cmd[0] == "mount" and cmd[2] == self.fail_on:
|
||||||
|
R.returncode = 32
|
||||||
|
R.stderr = "mount failed"
|
||||||
|
return R
|
||||||
|
|
||||||
|
|
||||||
|
class TestApplyPlanRollback:
|
||||||
|
def test_rolls_back_mounts_when_one_fails(self, tmp_path):
|
||||||
|
# A half-built tree must never reach the backup tool.
|
||||||
|
root = str(tmp_path / "root")
|
||||||
|
mounts = [("/src", root), ("/src/a", root + "/a"), ("/src/b", root + "/b")]
|
||||||
|
|
||||||
|
runner = FakeRunner(fail_on="/src/b")
|
||||||
|
with pytest.raises(StagingError, match="bind-mount"):
|
||||||
|
apply_plan(mounts, runner=runner, isdir=lambda _p: True)
|
||||||
|
|
||||||
|
umounts = [c[-1] for c in runner.calls if c[0] == "umount"]
|
||||||
|
assert umounts == [root + "/a", root]
|
||||||
|
|
||||||
|
def test_raises_when_target_missing(self, tmp_path):
|
||||||
|
root = str(tmp_path / "root")
|
||||||
|
with pytest.raises(StagingError, match="does not exist"):
|
||||||
|
apply_plan(
|
||||||
|
[("/src", root), ("/src/a", root + "/a")],
|
||||||
|
runner=FakeRunner(),
|
||||||
|
isdir=lambda p: p == root,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class TestTeardown:
|
||||||
|
def test_unmounts_deepest_first(self, tmp_path):
|
||||||
|
mounts_file = tmp_path / "mounts"
|
||||||
|
mounts_file.write_text(
|
||||||
|
f"tmpfs {ROOT} tmpfs rw 0 0\n"
|
||||||
|
f"tmpfs {ROOT}/apps tmpfs rw 0 0\n"
|
||||||
|
f"tmpfs {ROOT}/apps/lidarr/config tmpfs rw 0 0\n"
|
||||||
|
f"tmpfs {ROOT}/apps/lidarr tmpfs rw 0 0\n"
|
||||||
|
"tmpfs /somewhere/else tmpfs rw 0 0\n"
|
||||||
|
)
|
||||||
|
runner = FakeRunner()
|
||||||
|
teardown(ROOT, runner=runner, mounts_file=str(mounts_file))
|
||||||
|
|
||||||
|
order = [c[-1] for c in runner.calls if c[0] == "umount"]
|
||||||
|
assert order == [
|
||||||
|
f"{ROOT}/apps/lidarr/config",
|
||||||
|
f"{ROOT}/apps/lidarr",
|
||||||
|
f"{ROOT}/apps",
|
||||||
|
ROOT,
|
||||||
|
]
|
||||||
|
assert "/somewhere/else" not in order
|
||||||
|
|
||||||
|
def test_is_idempotent_when_nothing_mounted(self, tmp_path):
|
||||||
|
mounts_file = tmp_path / "mounts"
|
||||||
|
mounts_file.write_text("tmpfs /somewhere/else tmpfs rw 0 0\n")
|
||||||
|
runner = FakeRunner()
|
||||||
|
assert teardown(ROOT, runner=runner, mounts_file=str(mounts_file)) == []
|
||||||
|
assert runner.calls == []
|
||||||
|
|
||||||
|
def test_falls_back_to_lazy_umount(self, tmp_path):
|
||||||
|
mounts_file = tmp_path / "mounts"
|
||||||
|
mounts_file.write_text(f"tmpfs {ROOT} tmpfs rw 0 0\n")
|
||||||
|
|
||||||
|
class Busy(FakeRunner):
|
||||||
|
def __call__(self, cmd):
|
||||||
|
self.calls.append(cmd)
|
||||||
|
|
||||||
|
class R:
|
||||||
|
returncode = 0 if "-l" in cmd else 32
|
||||||
|
stderr = "target is busy"
|
||||||
|
|
||||||
|
return R
|
||||||
|
|
||||||
|
runner = Busy()
|
||||||
|
assert teardown(ROOT, runner=runner, mounts_file=str(mounts_file)) == []
|
||||||
|
assert ["umount", "-l", ROOT] in runner.calls
|
||||||
|
|
||||||
|
|
||||||
|
class TestCleanupAll:
|
||||||
|
"""uninstall.sh and recover.sh call this instead of reimplementing teardown."""
|
||||||
|
|
||||||
|
def test_reports_orphan_snapshots_before_deleting_their_sidecars(self, tmp_path):
|
||||||
|
# The sidecar is the only record that an interrupted run's snapshot tree
|
||||||
|
# is still on disk. Deleting it without naming the snapshot orphans the
|
||||||
|
# whole tree silently.
|
||||||
|
base = tmp_path / "stage"
|
||||||
|
base.mkdir()
|
||||||
|
(base / "cloud_backup-5.snapshot").write_text("Tap@interrupted")
|
||||||
|
|
||||||
|
mounts_file = tmp_path / "mounts"
|
||||||
|
mounts_file.write_text("")
|
||||||
|
|
||||||
|
lines, errors = cleanup_all(
|
||||||
|
base=str(base), runner=FakeRunner(), mounts_file=str(mounts_file)
|
||||||
|
)
|
||||||
|
assert errors == []
|
||||||
|
assert any("Tap@interrupted" in ln for ln in lines)
|
||||||
|
assert any("zfs destroy -r" in ln for ln in lines)
|
||||||
|
# Sidecar cleared only after being reported.
|
||||||
|
assert not (base / "cloud_backup-5.snapshot").exists()
|
||||||
|
|
||||||
|
def test_unmounts_everything_under_the_base_deepest_first(self, tmp_path):
|
||||||
|
base = tmp_path / "stage"
|
||||||
|
base.mkdir()
|
||||||
|
mounts_file = tmp_path / "mounts"
|
||||||
|
mounts_file.write_text(
|
||||||
|
f"tmpfs {base} tmpfs rw 0 0\n"
|
||||||
|
f"tmpfs {base}/cloud_backup-5 tmpfs rw 0 0\n"
|
||||||
|
f"tmpfs {base}/cloud_backup-5/apps tmpfs rw 0 0\n"
|
||||||
|
)
|
||||||
|
runner = FakeRunner()
|
||||||
|
_lines, errors = cleanup_all(
|
||||||
|
base=str(base), runner=runner, mounts_file=str(mounts_file)
|
||||||
|
)
|
||||||
|
assert errors == []
|
||||||
|
order = [c[-1] for c in runner.calls if c[0] == "umount"]
|
||||||
|
assert order == [
|
||||||
|
f"{base}/cloud_backup-5/apps",
|
||||||
|
f"{base}/cloud_backup-5",
|
||||||
|
str(base),
|
||||||
|
]
|
||||||
|
|
||||||
|
def test_keeps_sidecars_when_an_unmount_failed(self, tmp_path):
|
||||||
|
# If a mount is stuck, the snapshot is still pinned — so the record of it
|
||||||
|
# must survive for the next run (or the operator) to act on.
|
||||||
|
base = tmp_path / "stage"
|
||||||
|
base.mkdir()
|
||||||
|
(base / "cloud_backup-5.snapshot").write_text("Tap@stuck")
|
||||||
|
mounts_file = tmp_path / "mounts"
|
||||||
|
mounts_file.write_text(f"tmpfs {base}/cloud_backup-5 tmpfs rw 0 0\n")
|
||||||
|
|
||||||
|
class Stuck(FakeRunner):
|
||||||
|
def __call__(self, cmd):
|
||||||
|
self.calls.append(cmd)
|
||||||
|
|
||||||
|
class R:
|
||||||
|
returncode = 32
|
||||||
|
stderr = "target is busy"
|
||||||
|
|
||||||
|
return R
|
||||||
|
|
||||||
|
_lines, errors = cleanup_all(
|
||||||
|
base=str(base), runner=Stuck(), mounts_file=str(mounts_file)
|
||||||
|
)
|
||||||
|
assert errors, "a stuck unmount must be reported"
|
||||||
|
assert (base / "cloud_backup-5.snapshot").exists()
|
||||||
|
|
||||||
|
def test_is_a_noop_on_a_clean_system(self, tmp_path):
|
||||||
|
mounts_file = tmp_path / "mounts"
|
||||||
|
mounts_file.write_text("")
|
||||||
|
lines, errors = cleanup_all(
|
||||||
|
base=str(tmp_path / "absent"), runner=FakeRunner(),
|
||||||
|
mounts_file=str(mounts_file),
|
||||||
|
)
|
||||||
|
assert errors == []
|
||||||
|
assert lines == [" None active."]
|
||||||
|
|
||||||
|
|
||||||
|
class TestCurrentMountsUnder:
|
||||||
|
def test_matches_only_the_staging_subtree(self, tmp_path):
|
||||||
|
mounts_file = tmp_path / "mounts"
|
||||||
|
# "cloud_backup-50" must NOT match "cloud_backup-5".
|
||||||
|
mounts_file.write_text(
|
||||||
|
f"tmpfs {ROOT} tmpfs rw 0 0\n"
|
||||||
|
"tmpfs /run/truecloud-nested/cloud_backup-50 tmpfs rw 0 0\n"
|
||||||
|
)
|
||||||
|
assert current_mounts_under(ROOT, mounts_file=str(mounts_file)) == [ROOT]
|
||||||
|
|
||||||
|
|
||||||
|
class TestStagingRootFor:
|
||||||
|
def test_stable_per_task(self):
|
||||||
|
assert staging_root_for("cloud_backup-5") == "/run/truecloud-nested/cloud_backup-5"
|
||||||
|
|
||||||
|
def test_sanitises_path_separators(self):
|
||||||
|
assert "/" not in staging_root_for("evil/name").rsplit("/", 1)[-1]
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("name", ["..", ".", "...", "/", ""])
|
||||||
|
def test_dot_components_cannot_escape_the_staging_base(self, name):
|
||||||
|
# os.path.join(BASE, "..") normalises to /run — teardown would rmdir it.
|
||||||
|
root = staging_root_for(name)
|
||||||
|
assert os.path.normpath(root).startswith("/run/truecloud-nested/")
|
||||||
@@ -0,0 +1,157 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Extract one version's section from CHANGELOG.md, and check version consistency.
|
||||||
|
|
||||||
|
Used by .github/workflows/release.yml so a release's body is always the changelog
|
||||||
|
entry -- there is no second place to write release notes, and therefore no second
|
||||||
|
place for them to be wrong.
|
||||||
|
|
||||||
|
python3 tools/release_notes.py notes v0.3.0 # -> the section body
|
||||||
|
python3 tools/release_notes.py version # -> version per the scripts
|
||||||
|
python3 tools/release_notes.py check v0.3.0 # -> exit 1 on any mismatch
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
|
||||||
|
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||||
|
CHANGELOG = os.path.join(ROOT, "CHANGELOG.md")
|
||||||
|
|
||||||
|
# Every script prints a version; they must all agree, and agree with the tag.
|
||||||
|
# They drifted to three different values once (0.0.4 / 0.2.1) before anything
|
||||||
|
# checked them.
|
||||||
|
VERSIONED_FILES = [
|
||||||
|
"install.sh",
|
||||||
|
"uninstall.sh",
|
||||||
|
"recover.sh",
|
||||||
|
os.path.join("patch", "apply.sh"),
|
||||||
|
]
|
||||||
|
|
||||||
|
_VERSION_RE = re.compile(r'^VERSION="([^"]+)"', re.M)
|
||||||
|
_HEADING_RE = re.compile(r"^##\s+v?(\d+\.\d+\.\d+[^\s]*)", re.M)
|
||||||
|
|
||||||
|
|
||||||
|
def normalise(v: str) -> str:
|
||||||
|
return v.strip().lstrip("v")
|
||||||
|
|
||||||
|
|
||||||
|
def script_versions(root: str = ROOT) -> dict[str, str]:
|
||||||
|
"""VERSION= as declared by each script."""
|
||||||
|
found = {}
|
||||||
|
for rel in VERSIONED_FILES:
|
||||||
|
path = os.path.join(root, rel)
|
||||||
|
try:
|
||||||
|
with open(path, encoding="utf-8") as fh:
|
||||||
|
m = _VERSION_RE.search(fh.read())
|
||||||
|
except OSError:
|
||||||
|
continue
|
||||||
|
if m:
|
||||||
|
found[rel] = m.group(1)
|
||||||
|
return found
|
||||||
|
|
||||||
|
|
||||||
|
def changelog_versions(text: str) -> list[str]:
|
||||||
|
"""Versions with a section in the changelog, newest first."""
|
||||||
|
return [normalise(v) for v in _HEADING_RE.findall(text)]
|
||||||
|
|
||||||
|
|
||||||
|
def extract_notes(text: str, version: str) -> str:
|
||||||
|
"""The body of one version's section, without its heading.
|
||||||
|
|
||||||
|
Raises KeyError if the version has no section -- a release with an empty or
|
||||||
|
wrong body is worse than a failed release.
|
||||||
|
"""
|
||||||
|
want = normalise(version)
|
||||||
|
lines = text.splitlines()
|
||||||
|
|
||||||
|
start = None
|
||||||
|
for i, line in enumerate(lines):
|
||||||
|
m = _HEADING_RE.match(line)
|
||||||
|
if m and normalise(m.group(1)) == want:
|
||||||
|
start = i + 1
|
||||||
|
break
|
||||||
|
if start is None:
|
||||||
|
raise KeyError(f"CHANGELOG.md has no section for v{want}")
|
||||||
|
|
||||||
|
end = len(lines)
|
||||||
|
for i in range(start, len(lines)):
|
||||||
|
if _HEADING_RE.match(lines[i]):
|
||||||
|
end = i
|
||||||
|
break
|
||||||
|
|
||||||
|
return "\n".join(lines[start:end]).strip()
|
||||||
|
|
||||||
|
|
||||||
|
def check(version: str, root: str = ROOT) -> list[str]:
|
||||||
|
"""Every reason this version is not releasable. Empty list means it is."""
|
||||||
|
want = normalise(version)
|
||||||
|
problems = []
|
||||||
|
|
||||||
|
versions = script_versions(root)
|
||||||
|
for rel, got in sorted(versions.items()):
|
||||||
|
if normalise(got) != want:
|
||||||
|
problems.append(f"{rel} declares VERSION={got!r}, tag is v{want}")
|
||||||
|
missing = [r for r in VERSIONED_FILES if r not in versions]
|
||||||
|
for rel in missing:
|
||||||
|
problems.append(f"{rel} has no VERSION= line")
|
||||||
|
|
||||||
|
try:
|
||||||
|
with open(os.path.join(root, "CHANGELOG.md"), encoding="utf-8") as fh:
|
||||||
|
text = fh.read()
|
||||||
|
except OSError as e:
|
||||||
|
problems.append(f"cannot read CHANGELOG.md: {e}")
|
||||||
|
return problems
|
||||||
|
|
||||||
|
try:
|
||||||
|
body = extract_notes(text, want)
|
||||||
|
except KeyError as e:
|
||||||
|
problems.append(str(e))
|
||||||
|
else:
|
||||||
|
if not body:
|
||||||
|
problems.append(f"CHANGELOG.md section for v{want} is empty")
|
||||||
|
|
||||||
|
return problems
|
||||||
|
|
||||||
|
|
||||||
|
def main(argv):
|
||||||
|
if len(argv) < 2:
|
||||||
|
print(__doc__, file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
|
||||||
|
cmd = argv[1]
|
||||||
|
|
||||||
|
if cmd == "version":
|
||||||
|
versions = set(map(normalise, script_versions().values()))
|
||||||
|
if len(versions) != 1:
|
||||||
|
print(f"scripts disagree on version: {sorted(versions)}", file=sys.stderr)
|
||||||
|
return 1
|
||||||
|
print(versions.pop())
|
||||||
|
return 0
|
||||||
|
|
||||||
|
if len(argv) < 3:
|
||||||
|
print(f"usage: {argv[0]} {cmd} <version>", file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
version = argv[2]
|
||||||
|
|
||||||
|
if cmd == "notes":
|
||||||
|
with open(CHANGELOG, encoding="utf-8") as fh:
|
||||||
|
print(extract_notes(fh.read(), version))
|
||||||
|
return 0
|
||||||
|
|
||||||
|
if cmd == "check":
|
||||||
|
problems = check(version)
|
||||||
|
for p in problems:
|
||||||
|
print(f"::error::{p}")
|
||||||
|
if problems:
|
||||||
|
return 1
|
||||||
|
print(f"v{normalise(version)} is consistent across scripts and CHANGELOG")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
print(f"unknown command: {cmd}", file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main(sys.argv))
|
||||||
+22
-1
@@ -3,7 +3,7 @@
|
|||||||
|
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
VERSION="0.0.4"
|
VERSION="0.3.2"
|
||||||
|
|
||||||
PATCH_DIR="$(cd "$(dirname "$0")" && pwd)"
|
PATCH_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||||
_HOOK_COMMENT='TrueCloud provider patch (S3/B2)'
|
_HOOK_COMMENT='TrueCloud provider patch (S3/B2)'
|
||||||
@@ -91,6 +91,27 @@ if [ "$_ov_found" -eq 0 ]; then
|
|||||||
fi
|
fi
|
||||||
echo ""
|
echo ""
|
||||||
|
|
||||||
|
# ── Unmount nested-snapshot staging trees ─────────────────────────────────────
|
||||||
|
# These bind mounts pin their ZFS snapshots, so they must go before anything
|
||||||
|
# tries to destroy those snapshots. Deepest first.
|
||||||
|
|
||||||
|
# Delegated to the patch module rather than reimplemented here: the depth
|
||||||
|
# ordering and lazy-umount fallback are fiddly, and a shell copy would be the
|
||||||
|
# untested one.
|
||||||
|
echo "Unmounting nested-snapshot staging trees (if any) ..."
|
||||||
|
if ! python3 "$PATCH_DIR/patch/truecloud_nested.py" cleanup; then
|
||||||
|
echo " WARNING: staging mounts remain. Unmount them manually; until you do,"
|
||||||
|
echo " the ZFS snapshots they pin cannot be destroyed."
|
||||||
|
fi
|
||||||
|
|
||||||
|
# The opt-in marker lives in the repo dir; remove it so a later re-install
|
||||||
|
# starts from the safe default (feature off).
|
||||||
|
if [ -f "$PATCH_DIR/nested_snapshots_enabled" ]; then
|
||||||
|
rm -f "$PATCH_DIR/nested_snapshots_enabled"
|
||||||
|
echo " Removed nested-snapshot opt-in marker."
|
||||||
|
fi
|
||||||
|
echo ""
|
||||||
|
|
||||||
if [ "$_restore_failed" -eq 1 ]; then
|
if [ "$_restore_failed" -eq 1 ]; then
|
||||||
echo ""
|
echo ""
|
||||||
echo "ERROR: One or more UI bundle backups could not be restored." >&2
|
echo "ERROR: One or more UI bundle backups could not be restored." >&2
|
||||||
|
|||||||
Reference in New Issue
Block a user