Compare commits
31
Commits
v0.4.0
...
v0.6.1-rc1
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c6b252ac6b | ||
|
|
841e0364fd | ||
|
|
0d04c2cd1c | ||
|
|
2b8ef107f7 | ||
|
|
b50567e9a7 | ||
|
|
1b2407f6e2 | ||
|
|
ab0b66c47d | ||
|
|
469a2e4651 | ||
|
|
518a22d87e | ||
|
|
8c1b4c45f5 | ||
|
|
c250bc8f5d | ||
|
|
1c46ef66b0 | ||
|
|
252696f1de | ||
|
|
a1e31e9c0a | ||
|
|
a54b4dd7f6 | ||
|
|
ecb64878ff | ||
|
|
498b2690e1 | ||
|
|
cf2c6a8a02 | ||
|
|
aa725fb198 | ||
|
|
ea090c7f72 | ||
|
|
f927773f81 | ||
|
|
cd39489c7f | ||
|
|
9236aa0034 | ||
|
|
5cbb7def6f | ||
|
|
4ced730d65 | ||
|
|
8a82bde531 | ||
|
|
0e9e22da18 | ||
|
|
bf6d37e621 | ||
|
|
4e0814028c | ||
|
|
345741e1f1 | ||
|
|
092bdeae29 |
@@ -0,0 +1,237 @@
|
||||
name: TrueNAS compatibility
|
||||
|
||||
# Find out that iX broke us BEFORE their release ships, not after a user's backup
|
||||
# fails.
|
||||
#
|
||||
# This patch appends code to middlewared's internal modules. There is no stability
|
||||
# contract: TrueNAS 26 rewrote the whole cloud_backup path from async to sync, and
|
||||
# every block the nested module injects is an `async def` wrapping an `await`ed
|
||||
# original. Nobody would have found out until a restore did not work.
|
||||
#
|
||||
# tools/compat.py records what the patch assumes and checks it against iX's actual
|
||||
# source at every release line -- including master and the current BETA/RC, which is
|
||||
# where a break shows up first. When an UNRELEASED line breaks, this opens a bug
|
||||
# report so there is time to fix it before that version reaches anyone.
|
||||
#
|
||||
# Runs on both forges: Gitea (canonical) and GitHub (mirror). Only the "file an
|
||||
# issue" call differs.
|
||||
|
||||
on:
|
||||
schedule:
|
||||
- cron: "17 6 * * *" # daily, off the hour: everyone crons on the hour
|
||||
workflow_dispatch:
|
||||
push:
|
||||
paths:
|
||||
# The manifest itself changed -- re-check immediately rather than waiting a day.
|
||||
- "tools/compat.py"
|
||||
- ".github/workflows/compat.yml"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
issues: write
|
||||
|
||||
jobs:
|
||||
compat:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.13"
|
||||
|
||||
# ONE pass over the network. Every other step renders from this JSON -- calling
|
||||
# compat.py three times would re-fetch every file from every release line three
|
||||
# times, and could even disagree with itself if iX pushed mid-run.
|
||||
#
|
||||
# The exit code is CAPTURED, not allowed to abort the step: a nonzero exit means
|
||||
# "a shipped release is broken", which is a result to report, not a reason to
|
||||
# die before reporting it. (Actions runs `bash -e`, so `cmd > out` followed by
|
||||
# `echo $?` never reaches the echo.) It becomes a job failure at the end, after
|
||||
# the bug report has been filed.
|
||||
- name: check every TrueNAS release line
|
||||
id: check
|
||||
run: |
|
||||
rc=0
|
||||
python3 tools/compat.py --matrix --json > /tmp/matrix.json || rc=$?
|
||||
echo "shipped_broken=$rc" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# The report body is written to a FILE, and never becomes a step output.
|
||||
#
|
||||
# An earlier version did `echo "${{ steps.report.outputs.body }}"`, which
|
||||
# splices the text into the shell script itself -- and the report is full of
|
||||
# backticks, so bash ran `create-snapshot`, `def` and `async` as commands. It
|
||||
# is also an injection vector: the report is built from iX's source, so
|
||||
# anything that lands in middleware would execute on the runner.
|
||||
#
|
||||
# The rule that avoids the whole class: never interpolate ${{ }} into a `run:`
|
||||
# body. Files for data, `env:` for scalars (the runner sets those, rather than
|
||||
# pasting them into the script).
|
||||
- name: build the report
|
||||
id: report
|
||||
run: |
|
||||
python3 - <<'PY' >> "$GITHUB_OUTPUT"
|
||||
import json, sys
|
||||
|
||||
sys.path.insert(0, "tools")
|
||||
import compat
|
||||
|
||||
with open("/tmp/matrix.json") as fh:
|
||||
rows = json.load(fh)
|
||||
|
||||
with open("/tmp/matrix.md", "w") as fh:
|
||||
fh.write(compat.render_markdown(rows))
|
||||
|
||||
broken = [
|
||||
r for r in rows
|
||||
if any(compat.is_broken(m) for m in r["modules"].values())
|
||||
]
|
||||
native = [
|
||||
(r["ref"], mod)
|
||||
for r in rows
|
||||
for mod, m in sorted(r["modules"].items())
|
||||
if m["native"] and not compat.is_broken(m)
|
||||
]
|
||||
|
||||
lines = [
|
||||
"`tools/compat.py` found that the patch's assumptions about "
|
||||
"middlewared no longer hold.",
|
||||
"",
|
||||
compat.render_markdown(rows),
|
||||
"",
|
||||
]
|
||||
for r in broken:
|
||||
lines.append(f"### {r['ref']}")
|
||||
lines.append("")
|
||||
for mod, m in sorted(r["modules"].items()):
|
||||
if not compat.is_broken(m):
|
||||
continue
|
||||
lines.append(f"**{mod}** — the patch will not apply:")
|
||||
lines.append("")
|
||||
for p in m["problems"]:
|
||||
lines.append(f"- `{p['id']}`: {p['detail']}")
|
||||
lines.append(f" - why it matters: {p['why']}")
|
||||
lines.append("")
|
||||
for ref, mod in native:
|
||||
lines.append(
|
||||
f"- `{ref}`: **{mod}** appears to be NATIVE now — retire the "
|
||||
f"module rather than fixing it."
|
||||
)
|
||||
lines += ["", "_Filed automatically by `.github/workflows/compat.yml`._"]
|
||||
|
||||
with open("/tmp/issue.md", "w") as fh:
|
||||
fh.write("\n".join(lines))
|
||||
|
||||
# Scalars only. The body stays in the file.
|
||||
print(f"broken={'1' if broken else '0'}")
|
||||
print(f"refs={','.join(r['ref'] for r in broken)}")
|
||||
PY
|
||||
|
||||
- name: matrix
|
||||
run: cat /tmp/matrix.md
|
||||
|
||||
# Keep the README's table true. A support matrix that quietly goes stale is not
|
||||
# a stale doc -- it is a false promise to somebody deciding whether to trust
|
||||
# this with their backups.
|
||||
#
|
||||
# Only ever touches the block between the COMPAT MATRIX markers, and only on
|
||||
# the canonical host (Gitea) so the two forges cannot race each other. The
|
||||
# `paths:` trigger above does not include README.md, so this cannot re-trigger
|
||||
# itself; and a README change is documentation-only, which by design raises no
|
||||
# update alert on anyone's box.
|
||||
- name: refresh the README matrix
|
||||
if: ${{ github.event_name == 'schedule' && !contains(github.server_url, 'github.com') }}
|
||||
run: |
|
||||
python3 - <<'PY'
|
||||
import json, sys
|
||||
sys.path.insert(0, "tools")
|
||||
import compat
|
||||
with open("/tmp/matrix.json") as fh:
|
||||
rows = json.load(fh)
|
||||
print("changed" if compat.update_readme(rows) else "unchanged")
|
||||
PY
|
||||
|
||||
if ! git diff --quiet -- README.md; then
|
||||
git config user.name "truecloud-patch bot"
|
||||
git config user.email "bot@onetick.ninja"
|
||||
git add README.md
|
||||
git commit -m "docs: refresh the TrueNAS compatibility matrix"
|
||||
git push origin HEAD:main
|
||||
fi
|
||||
|
||||
# A broken SHIPPED release is an outage: users are on it right now.
|
||||
- name: fail if a shipped release is broken
|
||||
if: ${{ steps.check.outputs.shipped_broken != '0' }}
|
||||
run: |
|
||||
echo "::error::The patch is broken on a SHIPPED TrueNAS release."
|
||||
exit 1
|
||||
|
||||
- name: file a bug report (GitHub)
|
||||
if: ${{ steps.report.outputs.broken == '1' && contains(github.server_url, 'github.com') }}
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
TITLE: "TrueNAS compatibility: the patch's assumptions no longer hold"
|
||||
run: |
|
||||
# One issue per set of broken refs, reopened/updated rather than duplicated
|
||||
# daily -- a bot that files the same issue every morning gets muted, and
|
||||
# then it is not a warning system any more.
|
||||
# Lowest-numbered match, for the same reason as the Gitea step below.
|
||||
existing="$(gh issue list --state all --search "$TITLE" \
|
||||
--json number,title \
|
||||
--jq '[.[] | select(.title == env.TITLE) | .number] | min // empty')"
|
||||
|
||||
if [ -n "$existing" ]; then
|
||||
gh issue comment "$existing" --body-file /tmp/issue.md
|
||||
gh issue reopen "$existing" 2>/dev/null || true
|
||||
else
|
||||
gh issue create --title "$TITLE" --body-file /tmp/issue.md
|
||||
fi
|
||||
|
||||
- name: file a bug report (Gitea)
|
||||
if: ${{ steps.report.outputs.broken == '1' && !contains(github.server_url, 'github.com') }}
|
||||
env:
|
||||
TOKEN: ${{ secrets.GITEA_TOKEN || github.token }}
|
||||
API: ${{ github.server_url }}/api/v1/repos/${{ github.repository }}
|
||||
TITLE: "TrueNAS compatibility: the patch's assumptions no longer hold"
|
||||
run: |
|
||||
# python3, not jq: jq is not guaranteed on a self-hosted runner, and a bug
|
||||
# report that dies on a missing tool is a warning system that does not warn.
|
||||
python3 - <<'PY'
|
||||
import json, os, urllib.error, urllib.request
|
||||
|
||||
api, token, title = os.environ["API"], os.environ["TOKEN"], os.environ["TITLE"]
|
||||
with open("/tmp/issue.md", encoding="utf-8") as fh:
|
||||
body = fh.read()
|
||||
headers = {"Authorization": f"token {token}",
|
||||
"Content-Type": "application/json"}
|
||||
|
||||
def call(url, method, data=None):
|
||||
req = urllib.request.Request(
|
||||
url, method=method, headers=headers,
|
||||
data=json.dumps(data).encode() if data else None)
|
||||
with urllib.request.urlopen(req) as r: # noqa: S310
|
||||
return json.load(r) if r.length != 0 else {}
|
||||
|
||||
# Same title => same issue. Comment on it rather than filing a new one every
|
||||
# morning: a bot that duplicates itself daily gets muted, and then it is not
|
||||
# a warning system any more.
|
||||
# LOWEST-numbered match, not "whichever the API returns first". Two issues
|
||||
# with the same title already existed once (the old title embedded the ref
|
||||
# list, so the identity changed when that set changed), and an
|
||||
# order-dependent pick would have alternated between them, reopening one and
|
||||
# commenting on the other. Lowest number is stable no matter what the API
|
||||
# sorts by.
|
||||
issues = call(f"{api}/issues?state=all&type=issues", "GET")
|
||||
matches = sorted((i for i in issues if i["title"] == title),
|
||||
key=lambda i: i["number"])
|
||||
match = matches[0] if matches else None
|
||||
|
||||
if match:
|
||||
n = match["number"]
|
||||
call(f"{api}/issues/{n}/comments", "POST", {"body": body})
|
||||
call(f"{api}/issues/{n}", "PATCH", {"state": "open"})
|
||||
print(f"commented on and reopened issue #{n}")
|
||||
else:
|
||||
made = call(f"{api}/issues", "POST", {"title": title, "body": body})
|
||||
print(f"filed issue #{made['number']}")
|
||||
PY
|
||||
+119
-13
@@ -30,15 +30,30 @@ jobs:
|
||||
release:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
# Via `env:`, never spliced into the script. `inputs.tag` is attacker-chosen on
|
||||
# a workflow_dispatch, and a ${{ }} in a `run:` body is pasted into the shell
|
||||
# TEXT -- a tag of `$(...)` would simply execute. env: is safe: the runner sets
|
||||
# the variable instead of rewriting the script.
|
||||
- name: Resolve tag
|
||||
id: tag
|
||||
env:
|
||||
EVENT: ${{ github.event_name }}
|
||||
INPUT_TAG: ${{ inputs.tag }}
|
||||
run: |
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
echo "tag=${{ inputs.tag }}" >> "$GITHUB_OUTPUT"
|
||||
if [ "$EVENT" = "workflow_dispatch" ]; then
|
||||
tag="$INPUT_TAG"
|
||||
else
|
||||
echo "tag=${GITHUB_REF#refs/tags/}" >> "$GITHUB_OUTPUT"
|
||||
tag="${GITHUB_REF#refs/tags/}"
|
||||
fi
|
||||
|
||||
# Whatever it came from, it has to look like a tag we cut.
|
||||
case "$tag" in
|
||||
v[0-9]*.[0-9]*.[0-9]*) ;;
|
||||
*) echo "::error::refusing to release a tag that is not vX.Y.Z[-rcN]: $tag"; exit 1 ;;
|
||||
esac
|
||||
|
||||
echo "tag=$tag" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
ref: ${{ steps.tag.outputs.tag }}
|
||||
@@ -70,23 +85,62 @@ jobs:
|
||||
|
||||
# Catches the failure mode this repo actually had: VERSION= drifted to
|
||||
# three different values across the scripts, and nothing noticed.
|
||||
- name: version matches tag and CHANGELOG has a section
|
||||
run: python3 tools/release_notes.py check "${{ steps.tag.outputs.tag }}"
|
||||
- name: "gate: version matches tag, CHANGELOG complete, nothing stranded"
|
||||
env:
|
||||
TAG: ${{ steps.tag.outputs.tag }}
|
||||
run: python3 tools/release_notes.py check "$TAG"
|
||||
|
||||
# THE BARRIER. A stable release must have been a release candidate on this
|
||||
# exact commit. Candidates are invisible to users (update.sh and the alert
|
||||
# source both take the newest plain vX.Y.Z), so debugging happens across
|
||||
# rc1/rc2/rc3 at no cost to anyone -- instead of across v0.5.0/v0.5.1/v0.5.2,
|
||||
# which alerts every installed box every time.
|
||||
#
|
||||
# Same code release.sh runs locally, so this should never be the first place
|
||||
# you find out. It is here because this is the only place that cannot be
|
||||
# bypassed: it holds the token that publishes.
|
||||
- name: "gate: this commit was a release candidate"
|
||||
env:
|
||||
TAG: ${{ steps.tag.outputs.tag }}
|
||||
run: python3 tools/release_gate.py "$TAG" -C .
|
||||
|
||||
# There is deliberately NO "did the candidate's CI run pass?" gate here.
|
||||
#
|
||||
# It would have to query the forge's run history, which is the one thing that
|
||||
# differs between GitHub and Gitea -- and it adds nothing: the steps above
|
||||
# re-run ruff, pytest and the shell checks against the TAGGED COMMIT, and
|
||||
# release_gate.py has already proved a candidate points at that same commit.
|
||||
# If the code passes now, it passed then; they are the same code.
|
||||
#
|
||||
# What a candidate really buys is the thing no CI can check: that a human
|
||||
# installed it on a real box and exercised it. The barrier makes room for
|
||||
# that; it cannot verify it.
|
||||
|
||||
- name: extract release notes from CHANGELOG
|
||||
env:
|
||||
TAG: ${{ steps.tag.outputs.tag }}
|
||||
run: |
|
||||
python3 tools/release_notes.py notes "${{ steps.tag.outputs.tag }}" > /tmp/notes.md
|
||||
python3 tools/release_notes.py notes "$TAG" > /tmp/notes.md
|
||||
echo "--- release body ---"
|
||||
cat /tmp/notes.md
|
||||
|
||||
- name: create or update the release
|
||||
# This repo is canonically hosted on Gitea (git.onetick.ninja) and mirrored to
|
||||
# GitHub, and BOTH run this workflow -- Gitea reads .github/workflows too. So
|
||||
# the publish step has to work on whichever forge it lands on. Everything
|
||||
# above is forge-agnostic; only the "create a release" API differs.
|
||||
|
||||
- name: publish the release (GitHub)
|
||||
if: ${{ contains(github.server_url, 'github.com') }}
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
TAG: ${{ steps.tag.outputs.tag }}
|
||||
run: |
|
||||
# Pre-1.0 and any -rc/-beta suffix ship as prereleases, not "Latest".
|
||||
# Lowercased: release_gate/release_notes match the suffix case-INsensitively
|
||||
# (`is_prerelease` uses re.I), so a `v0.6.0-RC1` skipped the barrier as a
|
||||
# candidate and then landed here as a case-sensitive MISS -- published as the
|
||||
# forge's "Latest release" on a commit that was never a candidate.
|
||||
prerelease=""
|
||||
case "$TAG" in
|
||||
case "$(printf '%s' "$TAG" | tr '[:upper:]' '[:lower:]')" in
|
||||
*-rc*|*-beta*|*-alpha*) prerelease="--prerelease" ;;
|
||||
esac
|
||||
|
||||
@@ -95,8 +149,60 @@ jobs:
|
||||
gh release edit "$TAG" --notes-file /tmp/notes.md
|
||||
else
|
||||
# shellcheck disable=SC2086
|
||||
gh release create "$TAG" \
|
||||
--title "$TAG" \
|
||||
--notes-file /tmp/notes.md \
|
||||
$prerelease
|
||||
gh release create "$TAG" --title "$TAG" --notes-file /tmp/notes.md $prerelease
|
||||
fi
|
||||
|
||||
- name: publish the release (Gitea)
|
||||
if: ${{ !contains(github.server_url, 'github.com') }}
|
||||
env:
|
||||
TOKEN: ${{ secrets.GITEA_TOKEN || github.token }}
|
||||
TAG: ${{ steps.tag.outputs.tag }}
|
||||
API: ${{ github.server_url }}/api/v1/repos/${{ github.repository }}
|
||||
run: |
|
||||
prerelease=false
|
||||
case "$(printf '%s' "$TAG" | tr '[:upper:]' '[:lower:]')" in
|
||||
*-rc*|*-beta*|*-alpha*) prerelease=true ;;
|
||||
esac
|
||||
|
||||
# python3, not jq. The changelog is full of quotes, backticks and newlines,
|
||||
# so the body must be properly JSON-encoded -- but `jq` is not guaranteed on
|
||||
# a self-hosted Gitea runner, and a publish step that dies on a missing tool
|
||||
# leaves a tag with no release behind it. python3 is guaranteed: setup-python
|
||||
# ran above.
|
||||
python3 - "$TAG" "$API" "$TOKEN" "$prerelease" <<'PY'
|
||||
import json, sys, urllib.error, urllib.request
|
||||
|
||||
tag, api, token, prerelease = sys.argv[1:5]
|
||||
with open("/tmp/notes.md", encoding="utf-8") as fh:
|
||||
body = fh.read()
|
||||
|
||||
payload = {
|
||||
"tag_name": tag, "name": tag, "body": body,
|
||||
"prerelease": prerelease == "true",
|
||||
}
|
||||
headers = {
|
||||
"Authorization": f"token {token}",
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
|
||||
def call(url, method, data=None):
|
||||
req = urllib.request.Request(
|
||||
url, method=method, headers=headers,
|
||||
data=json.dumps(data).encode() if data else None)
|
||||
with urllib.request.urlopen(req) as r: # noqa: S310
|
||||
return r.status, json.load(r) if r.length != 0 else {}
|
||||
|
||||
try:
|
||||
_, existing = call(f"{api}/releases/tags/{tag}", "GET")
|
||||
except urllib.error.HTTPError as e:
|
||||
if e.code != 404:
|
||||
raise
|
||||
existing = None
|
||||
|
||||
if existing:
|
||||
status, _ = call(f"{api}/releases/{existing['id']}", "PATCH", payload)
|
||||
print(f"updated release {tag} -> {status}")
|
||||
else:
|
||||
status, _ = call(f"{api}/releases", "POST", payload)
|
||||
print(f"created release {tag} (prerelease={payload['prerelease']}) -> {status}")
|
||||
PY
|
||||
|
||||
@@ -5,6 +5,9 @@
|
||||
/hook_status.json
|
||||
/disabled
|
||||
/nested_snapshots_enabled
|
||||
# Written by apply.sh when a module's assumptions no longer fit the installed
|
||||
# middlewared (see tools/compat.py). Evidence for the alert; not source.
|
||||
/incompatible.json
|
||||
|
||||
# Python
|
||||
__pycache__/
|
||||
@@ -13,3 +16,4 @@ __pycache__/
|
||||
.ruff_cache/
|
||||
.venv/
|
||||
venv/
|
||||
/update_alerts_disabled
|
||||
|
||||
+384
@@ -1,5 +1,389 @@
|
||||
# Changelog
|
||||
|
||||
Work lands under **Unreleased** and stays there until a release promotes it. That
|
||||
is deliberate: see [Releasing](docs/releasing.md). Twelve releases were cut on
|
||||
2026-07-13, several of them fixing the release before — and with the update alert
|
||||
live, every one of those interrupts every user. An alert people learn to ignore is
|
||||
worse than no alert, because one day it carries a security fix.
|
||||
|
||||
## v0.6.1 — 2026-07-13
|
||||
### Fixed
|
||||
|
||||
- **A reboot mid-backup orphaned the entire snapshot tree, permanently.** The sidecar
|
||||
is the record of which snapshots a run pinned — and it lives in `/run`, which is
|
||||
**tmpfs**. A reboot (or a crash) between taking the recursive snapshot and cleaning
|
||||
it up destroyed that record, leaving one snapshot per descendant dataset — **250+ on
|
||||
a real pool** — with nothing left pointing at them. Nothing would ever have found
|
||||
them again.
|
||||
|
||||
`gc_stale_snapshots()` is the backstop: it identifies leftovers **by name**, so it
|
||||
works when the record is gone. It runs at the start of every backup, after the
|
||||
sidecar reclaim — the recorded path stays authoritative, and the collector only ever
|
||||
mops up what the record lost.
|
||||
|
||||
Because it deletes data on a *name match* — a weaker claim than a recorded fact — the
|
||||
selection is a **pure function** with the harshest tests in the suite. A snapshot is
|
||||
collected only if **all** of these hold:
|
||||
|
||||
| | |
|
||||
| --- | --- |
|
||||
| name is exactly `<dataset>@<task>-<YYYYMMDDHHMMSS>` | so `cloud_backup-5` never matches `cloud_backup-50`, an `auto-*` periodic snapshot, or anything a human made |
|
||||
| it is not the current run's | parent *and* children are excluded |
|
||||
| **nothing is mounted from it** | an in-flight run pins its own snapshots — this, not the age guard, is what protects a concurrent backup |
|
||||
| it is **over an hour old** | covers the seconds-long window where a live run has snapshotted but not yet mounted |
|
||||
|
||||
Verified against the real pool: of **4,728** snapshots — including **2,341** periodic
|
||||
ones — it selects exactly the orphans of the task being run, and nothing else.
|
||||
|
||||
## v0.6.0 — 2026-07-13
|
||||
### Added
|
||||
|
||||
- **`release.sh` — a two-stage release process, and a barrier that enforces it.**
|
||||
A stable `vX.Y.Z` tag is now only publishable if a `vX.Y.Z-rcN` tag points at the
|
||||
**same commit**, and the release job re-runs the entire suite against that tagged
|
||||
commit before publishing. Candidates are invisible to users — `update.sh` and the
|
||||
update alert both take the newest plain `vX.Y.Z` tag — so debugging happens across
|
||||
rc1, rc2, rc3 at nobody's expense, instead of across v0.5.0, v0.5.1, v0.5.2 at
|
||||
everybody's.
|
||||
|
||||
bash release.sh 0.6.0 --rc # candidate. Invisible to users.
|
||||
bash release.sh 0.6.0 --promote # stable. Refused unless an rc passed HERE.
|
||||
|
||||
The rule is enforced in `tools/release_gate.py`, which `release.sh` runs locally
|
||||
(so you fail in 200 ms) and `.github/workflows/release.yml` runs again where it
|
||||
cannot be bypassed (so failing locally is not optional). "The candidate passed,
|
||||
then I pushed one more little fix" is refused by name — that is precisely how
|
||||
v0.5.1 happened.
|
||||
|
||||
- **TrueNAS compatibility is now checked, not hoped for.**
|
||||
[`tools/compat.py`](tools/compat.py) is a written-down record of everything each
|
||||
module assumes about middlewared, checked in two places:
|
||||
|
||||
- **CI, daily** — against iXsystems' source at every release line *including
|
||||
`master` and the current BETA/RC*. When an unreleased TrueNAS breaks the patch
|
||||
it files a bug report automatically, so there is time to fix it before that
|
||||
version reaches anyone. It also refreshes the README's support matrix, so the
|
||||
table cannot quietly become a false promise.
|
||||
- **`apply.sh`, at every boot** — against the middleware actually installed on the
|
||||
box. **A module whose assumptions no longer hold is not applied.** Stock TrueNAS
|
||||
without a feature beats TrueNAS with a broken backup.
|
||||
|
||||
It immediately found two real breaks: TrueNAS 26 (below), and a nested-snapshot bug
|
||||
that had been shipping for two releases (below).
|
||||
|
||||
- **The compatibility check now covers the middlewared methods the patch _calls_,**
|
||||
not only the symbols it wraps — and that gap was hiding a catastrophe.
|
||||
|
||||
TrueNAS 26 **deletes `plugins/zfs_/dataset.py` and `plugins/zfs_/snapshot.py`
|
||||
outright**, taking `zfs.dataset.query`, `zfs.snapshot.query` and
|
||||
`zfs.snapshot.delete` with them (26 uses `filesystem.statfs` and `zfs.resource.*`).
|
||||
Nothing about the five `cloud_backup` files reveals that, so every other check went
|
||||
green. The patch would have applied perfectly and then **failed on the first
|
||||
backup** — or, far worse, snapshotted successfully and failed to *delete*,
|
||||
orphaning one snapshot per descendant dataset (**250 on a real pool**) on every
|
||||
single run, forever.
|
||||
|
||||
This is now an assumption class of its own, so a method disappearing is a BROKEN
|
||||
verdict rather than a silent time bomb.
|
||||
|
||||
- **Groundwork for TrueNAS 26** (async→sync and the deleted helper — see below).
|
||||
**26 is still reported BROKEN and the nested module will not apply there**, because
|
||||
the ZFS API rewrite above is not yet ported. Porting it needs a real 26 box to
|
||||
verify against, and shipping a port nobody has run is exactly the failure this
|
||||
project exists to avoid. On 26, TrueNAS is left stock: B2/S3 keeps working, nested
|
||||
datasets are simply not covered.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **A few snapshots leaked on every nested run, forever.** Found on real hardware, in
|
||||
the one place it could be: a 256-snapshot backup of `/mnt/Tap` swept 253 cleanly and
|
||||
left **3 behind** with `dataset is busy`.
|
||||
|
||||
The cause is ZFS's own automount. Reading anything under
|
||||
`<dataset>/.zfs/snapshot/<snap>/` makes ZFS **automount that snapshot**, and it stays
|
||||
mounted for `zfs_expire_snapshot` seconds (**300** by default) after the last access.
|
||||
`teardown()` unmounts *our* bind mounts — but not the automount underneath — so
|
||||
`zfs destroy` refuses for exactly the datasets restic read most recently. Then
|
||||
`cleanup_task()` removed the sidecar anyway, destroying the only record that those
|
||||
snapshots existed. Nothing would ever have reclaimed them.
|
||||
|
||||
Three changes, and the third is the one that makes it safe rather than merely
|
||||
unlikely:
|
||||
- `release_snapdirs()` unmounts ZFS's own `.zfs/snapshot` automounts (deepest first)
|
||||
before deleting, so the snapshots are not busy in the first place.
|
||||
- `delete_snapshot_tree()` **retries** the transient busy, and **returns the
|
||||
snapshots it could not delete** instead of swallowing them.
|
||||
- **The sidecar is now removed only on a confirmed-clean sweep** — including on the
|
||||
staging-failure path, which used to remove it *before* the caller swept. The
|
||||
asymmetry is deliberate: a sidecar left behind when the tree is already gone costs
|
||||
one no-op delete on the next run, while a sidecar removed while the tree still
|
||||
exists is unrecoverable. Survivors are reclaimed by the next run.
|
||||
|
||||
**Expect the occasional straggler, and expect it to clean itself up.** On a
|
||||
256-snapshot tree this reliably sweeps ~255 immediately and may leave **one**: it is
|
||||
whatever restic read last, so its 300-second window has barely opened. That one is
|
||||
logged, its sidecar is kept, and the next run reclaims it before doing anything else.
|
||||
The leak is bounded at a single cycle rather than growing without limit — which is
|
||||
the property that actually matters. Blocking a backup job for five minutes to chase
|
||||
the last snapshot would be a worse trade, so it is not made.
|
||||
|
||||
- **Installing the patch permanently blocked updating it.** `install.sh` does
|
||||
`chmod +x update.sh`, and git recorded `update.sh` as `100644` — so the chmod was a
|
||||
*tracked modification*, and `update.sh` refuses to run over a dirty tree. Install
|
||||
once and you could never update again; the error even told you to run
|
||||
`git checkout -- .`, which just undoes the exec bit so the next install can re-dirty
|
||||
it. A real box sat on an old version for exactly this reason.
|
||||
|
||||
Fixed on both sides: the scripts `install.sh` chmods are now executable in git (so
|
||||
the chmod is a no-op), and `update.sh`'s dirty check now looks at **content**, not
|
||||
file mode — `git diff --numstat` reports `0 0` for a mode-only change. A test
|
||||
asserts every script in `install.sh`'s chmod loop is already `100755` in git.
|
||||
|
||||
- **Nested snapshots were broken on TrueNAS 24.10 and 25.04, and had been all
|
||||
along.** `SYNC_BLOCK`'s wrapper spelled out the stock signature and forwarded five
|
||||
arguments — but those releases declare `restic_backup(middleware, job,
|
||||
cloud_backup, dry_run)`; `rate_limit` only arrived in 25.10. Every nested backup on
|
||||
24.10/25.04 raised `TypeError: restic_backup() takes 4 positional arguments but 5
|
||||
were given`. The wrapper now takes `*args, **kwargs` and forwards whatever it is
|
||||
handed, so a trailing parameter appearing or disappearing is a non-event.
|
||||
|
||||
Found by the new compatibility check, not by a user — which is the whole argument
|
||||
for having it. The check it replaced only asked whether the parameter *names* still
|
||||
appeared somewhere in the signature, so it happily passed a call that could never
|
||||
work.
|
||||
|
||||
- **The nested module is now one synchronous implementation behind two thin
|
||||
wrappers.** TrueNAS 26 rewrites `cloud_backup` from async to **synchronous** and
|
||||
separately **deletes `get_dataset_recursive()`**, which an injected block called out
|
||||
of the host module's namespace. Either alone is a broken backup found at restore
|
||||
time: an `async def` wrapper hands `sync.py` a coroutine where it unpacks a tuple,
|
||||
and the vanished helper is a straight `NameError`.
|
||||
|
||||
The module now talks to middlewared through `call_sync`, and `apply.sh` reads which
|
||||
flavour the installed middleware declares and injects the matching wrapper —
|
||||
TrueNAS ≤ 25.10 reaches it via `await middleware.run_in_thread(...)`; a synchronous
|
||||
TrueNAS, already in a worker thread, calls it directly. The logic that owns the
|
||||
snapshots, the bind mounts and the failure modes exists **once**; an async twin
|
||||
would mean every future fix had to land twice, and the one that got missed would be
|
||||
the one that eats a backup. A middleware whose three wrapped functions **disagree**
|
||||
about async-ness is refused outright rather than guessed at, and
|
||||
`get_dataset_recursive` is carried as our own copy — removing the dependency on both
|
||||
versions instead of asserting it.
|
||||
|
||||
- **The patch no longer reaches into CloudSync tasks it has no business touching.**
|
||||
`create_snapshot` is module-global in `plugins/cloud/snapshot.py` and is imported by
|
||||
**`cloud_sync.py` as well as `cloud_backup/sync.py`** — so the wrapper sat in the
|
||||
path of every rclone/Storj **CloudSync** task with `snapshot=true`, and issued a
|
||||
`zfs.dataset.query` before deciding it had nothing to do. That added a brand-new
|
||||
failure mode to jobs that worked fine before this patch was installed, and worse: a
|
||||
CloudSync task that ever *did* get staged would **never be torn down**, because the
|
||||
teardown is wired into `cloud_backup`'s `restic_backup` and `CRUD_BLOCK`
|
||||
deliberately leaves CloudSync's nesting guard intact — the bind mounts would pin the
|
||||
ZFS snapshot forever. The staging path now bails out immediately unless the snapshot
|
||||
is named `cloud_backup-*`, before any middleware call.
|
||||
|
||||
- **Teardown warnings are no longer silently swallowed on TrueNAS ≤ 25.10.** The async
|
||||
wrapper's `finally` dropped the `logger=` kwarg that the sync one passes, so a
|
||||
cleanup that failed to unmount a bind mount *or* to delete a snapshot tree logged
|
||||
**nothing at all** — on the only platform anyone actually runs. `run_in_thread`
|
||||
forwards `**kwargs` via `functools.partial`; it was a regression, not a limitation.
|
||||
|
||||
- **`do_delete` is recognised as `delete`.** TrueNAS 24.10 and 25.04 declare
|
||||
`do_delete` (the `CRUDService` convention); 25.10 renamed it to `delete`. Both
|
||||
answer to `zfs.snapshot.delete`. Accepting only the literal name reported both older
|
||||
releases as BROKEN — a false verdict that would have switched nested snapshots off
|
||||
on boxes where they work perfectly.
|
||||
|
||||
- **An incompatible TrueNAS no longer sets the permanent kill switch.** `apply.sh`
|
||||
reused a "nothing left to do" exit that touches `disabled`, which suppresses
|
||||
patching on every future boot and is cleared only by `install.sh` — never by
|
||||
`update.sh`. On TrueNAS 26 (providers-compatible, nested opt-out by default) that
|
||||
branch would have fired, and the very release that fixed 26 could not have
|
||||
re-enabled itself: the user would run `bash update.sh`, exactly as the update alert
|
||||
tells them to, and the patch would stay dead with their B2 backups off.
|
||||
Incompatibility now means "apply nothing this boot, try again next boot".
|
||||
Retirement and incompatibility are opposite situations and no longer share an exit.
|
||||
|
||||
- **The compatibility check itself could be fooled**, in ways that each had teeth: a
|
||||
reordered, keyword-only, or newly-required parameter now reads as broken (the patch
|
||||
calls these positionally); a **re-exported or conditionally-defined** symbol reads
|
||||
as *unknown* rather than broken, so an innocent upstream refactor cannot make a
|
||||
working module decline to apply; an **unreadable** source (rate limit, DNS, timeout)
|
||||
is *unknown* rather than "iXsystems deleted this file", so a network blip cannot
|
||||
file a bug report, fail CI, and repaint the published support matrix; and `native`
|
||||
no longer masks `BROKEN`, which used to render a TrueNAS that both reworded the
|
||||
nesting guard *and* reshaped the functions as good news.
|
||||
|
||||
- **`compat.py --tree` no longer reads the patch's own code as native support.**
|
||||
`B2_BLOCK` writes `B2RcloneRemote.restic = True` into `b2.py` — exactly the string
|
||||
the providers native-probe looks for — so the one command the docs recommend for
|
||||
checking a live box said "retire the providers module" on every *patched* machine.
|
||||
It now reads only the part of the file iXsystems wrote.
|
||||
|
||||
- **`release.sh --promote` could never succeed.** It refused to run if the stable tag
|
||||
existed, and the gate refused if it did not — mutually exclusive, so the only way to
|
||||
cut a stable release was to hand-tag and bypass every gate this work exists to
|
||||
enforce. The gate now resolves the tag's commit if it exists and `HEAD` otherwise.
|
||||
The tests hid it by always tagging first.
|
||||
|
||||
### Changed
|
||||
|
||||
- **The minimum supported TrueNAS is stated, and enforced: 24.10.** TrueCloud Backup
|
||||
does not exist before it — `plugins/cloud_backup/` is simply absent — so the patch
|
||||
had nothing to attach to and would have done nothing at all, silently, while the
|
||||
user believed their backups were configured. `install.sh` now reads
|
||||
`system.version` and refuses, naming the reason. A version it cannot *parse* is a
|
||||
warning, not a refusal: declining to install over a string we failed to read would
|
||||
be a worse failure than the one being prevented.
|
||||
|
||||
- **A stable release may not leave work stranded under `## Unreleased`.** Either it
|
||||
is finished and belongs in the release, or the release is premature. Candidates
|
||||
are exempt: an rc may legitimately have work queued behind it.
|
||||
|
||||
- **`release.sh` refuses to run on an installed box.** The whole repo is cloned onto
|
||||
every box, so this file is there too; `update.sh` pins the checkout to a tag in
|
||||
detached HEAD, and `release.sh` now recognises that and says so, rather than
|
||||
emitting a confusing branch error.
|
||||
|
||||
- **Gitea (`git.onetick.ninja/flan/truenas-truecloud-patch`) is now canonical**, with
|
||||
GitHub as a mirror. Both forges run the same workflows and publish the same
|
||||
releases. The update alert now **derives the changelog URL from the `origin`
|
||||
remote** instead of hard-coding GitHub — which matters more than it sounds: when
|
||||
the changelog cannot be read, the alert deliberately fires *anyway* rather than risk
|
||||
hiding a security fix, so a stale URL would not have disabled the alert, it would
|
||||
have made it nag on every release, including documentation-only ones.
|
||||
|
||||
### Security
|
||||
|
||||
- **Workflow expressions are no longer interpolated into shell.**
|
||||
`echo "${{ steps.report.outputs.body }}"` pasted the compatibility report into the
|
||||
script text, and the report is full of backticks — bash ran `create-snapshot`,
|
||||
`def` and `async` as commands. Since that report is built from iXsystems' source,
|
||||
anything landing in their tree would have executed on the runner. `inputs.tag` on
|
||||
`workflow_dispatch` had the same shape, and that one is attacker-chosen. Data now
|
||||
moves through files and scalars through `env:`; a test enforces it across every
|
||||
workflow.
|
||||
|
||||
### Internal
|
||||
|
||||
- Static-analysis annotations in `patch/alert_source.py` (`# noqa` placement). No
|
||||
runtime change.
|
||||
|
||||
## v0.5.1 — 2026-07-13
|
||||
|
||||
### Fixed
|
||||
|
||||
- **The update alert could have broken middlewared at startup.** middlewared's
|
||||
`alert.load()` imports every file in `alert/source/` with **no try/except**, and
|
||||
it runs during setup — so a module that raises on import takes middlewared down
|
||||
with it. `apply.sh` now **compiles the substituted alert source and refuses to
|
||||
write it** if it does not parse. An uninstalled alert is a missing convenience;
|
||||
a broken one is a broken box.
|
||||
|
||||
- **`@PATCH_DIR@` is substituted with `repr()`**, so a repository path containing
|
||||
a quote or a backslash produces a valid Python literal instead of a syntax error
|
||||
in the installed module.
|
||||
|
||||
- **The alert source no longer mutates `sys.path`.** It loaded
|
||||
`tools/release_notes.py` via `sys.path.insert(0, …)`, which shadows the stdlib
|
||||
for that interpreter — and `ThreadedAlertSource` runs in middlewared's thread
|
||||
pool, so mutating `sys.path` is a race. It now loads the module by file path with
|
||||
`importlib`.
|
||||
|
||||
### Notes
|
||||
|
||||
Timing, for the record: `process_alerts` is `@periodic(60)` and
|
||||
`alert_source_last_run` is in-memory, so the check runs **within 60 seconds of any
|
||||
middlewared restart** (which this patch performs at every boot) and otherwise
|
||||
**within 24 hours** of a release.
|
||||
|
||||
## v0.5.0 — 2026-07-13
|
||||
|
||||
### Added
|
||||
|
||||
- **A TrueNAS alert when an update is available** — the bell in the UI, not a log
|
||||
line nobody reads. On by default, checked once a day.
|
||||
`install.sh --no-update-alerts` turns it off.
|
||||
|
||||
**It does not nag.** A release whose CHANGELOG contains only a `### Docs`
|
||||
section changed no code and raises nothing. Anything else raises INFO; a
|
||||
`### Security` section raises WARNING. The CHANGELOG's own section headings are
|
||||
the signal, and a security fix anywhere in the range escalates the whole span —
|
||||
so a docs-only release sitting on top of a security fix still reports as
|
||||
security, rather than hiding it.
|
||||
|
||||
**Why an AlertSource and not `midclt`:** TrueNAS cannot raise an alert from the
|
||||
CLI. `midclt` exposes only `alert.dismiss`, `alert.list`, `alert.list_categories`,
|
||||
`alert.list_policies` and `alert.restore` — alert *creation* is internal to
|
||||
middlewared, and none of its ~60 one-shot classes is generic enough to reuse. So
|
||||
registering an `AlertSource` is the only way, and it is also the least invasive
|
||||
thing this patch does: it **adds one file and modifies none**, where the
|
||||
providers and nested modules both append code to stock middleware files. It is
|
||||
the native mechanism, and TrueNAS polls it itself — no cron, no systemd timer.
|
||||
|
||||
- Fail-safe: every error path returns `None`; it cannot take middlewared down.
|
||||
- Read-only: `git ls-remote` plus an HTTPS fetch of the CHANGELOG. It never
|
||||
writes to `.git`, so it cannot leave root-owned objects behind the way a
|
||||
`git fetch` from middlewared (running as root) would.
|
||||
- Removed by `uninstall.sh`.
|
||||
- It only *tells* you; it never updates anything.
|
||||
|
||||
## v0.4.2 — 2026-07-13
|
||||
|
||||
### Docs
|
||||
|
||||
- **The Updating section never said how to *get* `update.sh`.** It ships inside the
|
||||
patch, so a clone older than v0.4.0 doesn't have it — the docs told you to run a
|
||||
script you didn't have. There is now an explicit bootstrap step (`git pull &&
|
||||
bash install.sh`, once), including the fix for the *"insufficient permission for
|
||||
adding an object to repository database"* failure that past `sudo git pull`s
|
||||
cause.
|
||||
|
||||
- **`After a TrueNAS update` rewritten.** It didn't explain that the patch
|
||||
re-applies itself at every boot (so you never reinstall), and it didn't say what
|
||||
each failure actually costs you. "Fail-safe" means *the box stays up* — not that
|
||||
your backups keep running. A `[FAIL] providers` is a **broken backup**, and the
|
||||
docs now say so rather than implying everything degrades gracefully.
|
||||
|
||||
- Added a repo map. `patch/mw_patch.py` and `tools/release_notes.py` were
|
||||
documented nowhere.
|
||||
|
||||
- `Development` told you to run `ruff check patch tests`, which misses `tools/`.
|
||||
|
||||
- Every command and file path in the README is now verified to exist and run.
|
||||
|
||||
## v0.4.1 — 2026-07-13
|
||||
|
||||
### Fixed
|
||||
|
||||
- **`update.sh` would have picked a release candidate as "the newest release".**
|
||||
Git's version sort ranks `v0.5.0-rc1` *above* `v0.5.0` (verified), and the
|
||||
release workflow deliberately supports rc/beta tags — so an RC would have been
|
||||
installed as though it were the latest stable. Tag selection is now filtered to
|
||||
plain `vX.Y.Z`.
|
||||
|
||||
- **`update.sh` would have died mid-update on an untracked file.** The dirty-tree
|
||||
guard uses `--untracked-files=no`, so an untracked file that the *target* tracks
|
||||
slipped past it — and `git checkout` then aborts. Under `set -e` the script died
|
||||
with a raw git error, *after* recording the rollback point. This is exactly what
|
||||
blocked a pull on a real box (a hand-copied `patch/wait_restart.sh`). It now
|
||||
detects the collision up front and names the files. Gitignored files are
|
||||
correctly *not* treated as blockers — git overwrites those silently.
|
||||
|
||||
Special case: if `update.sh` *itself* is the blocker, you hand-copied it in to
|
||||
bootstrap — and "delete `update.sh`, then re-run `update.sh`" is impossible. It
|
||||
now says so and prints the git commands that bootstrap it properly.
|
||||
|
||||
- **`--rollback` skipped that check entirely**, so it would have hit the identical
|
||||
failure. The check is now a shared function used by both paths, and rollback also
|
||||
validates that the recorded revision still exists (history can be rewritten).
|
||||
|
||||
- `install.sh`'s `chmod` aborted under `set -e` if any listed file was missing. The
|
||||
file set changes between versions, so `update.sh --rollback` to an older revision
|
||||
must not be killed by a filename this version happens to know about.
|
||||
|
||||
- `--to` with no value was silently ignored and fell back to the default target.
|
||||
|
||||
## v0.4.0 — 2026-07-13
|
||||
|
||||
### Added
|
||||
|
||||
@@ -1,407 +1,163 @@
|
||||
# truenas-truecloud-patch
|
||||
|
||||
Extends TrueNAS SCALE's **TrueCloud Backup** feature to:
|
||||
Extends TrueNAS SCALE's **TrueCloud Backup** to:
|
||||
|
||||
- work with S3-compatible providers and native Backblaze B2, instead of Storj only;
|
||||
- take **consistent snapshots of datasets that have child datasets** — which is
|
||||
every box running Apps (see [Nested-dataset snapshots](#nested-dataset-snapshots)).
|
||||
- back up to **Backblaze B2 and any S3-compatible provider**, not just Storj;
|
||||
- snapshot **datasets that have child datasets** — which is every box running Apps.
|
||||
|
||||
---
|
||||
**Requires TrueNAS SCALE 24.10 or newer.** TrueCloud Backup does not exist before
|
||||
that, and `install.sh` will refuse.
|
||||
|
||||
## Why this exists
|
||||
|
||||
In 2026, Storj raised the price of their TrueNAS-integrated storage tier from
|
||||
**$5/month to $50/month** — a 10× increase. For many home lab and small-office
|
||||
users, the TrueCloud Backup feature became unaffordable overnight.
|
||||
|
||||
TrueCloud Backup is the only native TrueNAS mechanism that provides:
|
||||
- Integrated ZFS snapshot support before each backup
|
||||
- Restic-based incremental deduplication
|
||||
- Scheduled tasks with progress and log tracking in the UI
|
||||
- Dataset lock integration
|
||||
|
||||
Running restic manually is possible but loses all of the above.
|
||||
This patch restores access to the TrueCloud Backup feature for users who
|
||||
need a provider other than Storj, with storage they already pay for or that
|
||||
costs a fraction of the new Storj price.
|
||||
|
||||
---
|
||||
|
||||
## Before you install
|
||||
|
||||
This project is unofficial and not affiliated with iXsystems. A few things worth
|
||||
knowing:
|
||||
|
||||
- It targets **internal middleware APIs** with no stability contract, so a
|
||||
TrueNAS update can break it. Every patch is fail-safe: if it can't apply,
|
||||
middlewared starts normally and the reason is logged to `apply.log`. Check the
|
||||
log after an update.
|
||||
- If you file a TrueNAS bug report, **remove the patch first** and reproduce on a
|
||||
stock system.
|
||||
- **Test your restores.** True of any backup, but it matters more here — see
|
||||
[Verifying it works](#verifying-it-works).
|
||||
- Provided as-is, no warranty. See LICENSE.
|
||||
|
||||
The patch is two independent modules — **providers** (B2/S3) and **nested**
|
||||
(snapshots on nested datasets) — and each retires on its own once TrueNAS ships
|
||||
that capability natively. See [Native support](#if-truenas-adds-native-support).
|
||||
|
||||
## Development
|
||||
|
||||
Parts of this project were written with AI assistance (Claude). All of it is
|
||||
reviewed and tested before release; the test suite and CI exist in large part to
|
||||
make that review meaningful. Bugs are mine.
|
||||
|
||||
```bash
|
||||
pip install pytest ruff
|
||||
ruff check patch tests
|
||||
pytest tests
|
||||
```
|
||||
|
||||
CI runs shellcheck, `bash -n`, ruff, and pytest on Python 3.11–3.13. The tests
|
||||
include a pass that `compile()`s the `*_BLOCK` strings in `patch/apply.sh` —
|
||||
those are Python source appended into live `middlewared` modules, so a syntax
|
||||
error there would break the box at boot.
|
||||
|
||||
---
|
||||
|
||||
## What is actually patched
|
||||
|
||||
**Nothing in TrueNAS's persistent database or configuration is modified**
|
||||
(other than the boot-hook entry itself). On every boot, `patch/apply.sh` runs
|
||||
as a PREINIT script. It mounts a writable
|
||||
[overlayfs](https://docs.kernel.org/filesystems/overlayfs.html) over the
|
||||
relevant directories in `/usr/` (upper layer in `/run` tmpfs), then patches
|
||||
`b2.py` and `restic.py` inside that overlay. The overlay is volatile — it
|
||||
exists only for the current boot — but the PREINIT script recreates it
|
||||
automatically on every subsequent boot. Nothing in `/usr/` is written to
|
||||
directly.
|
||||
|
||||
PREINIT scripts are executed *by* middlewared, which by then has already
|
||||
imported the stock modules — so after patching, `apply.sh` schedules a single
|
||||
detached middlewared restart (transient systemd unit `truecloud-mw-restart`
|
||||
running `patch/wait_restart.sh`) that loads the patched modules once boot has
|
||||
*actually* settled: the script waits for the systemd boot job queue to drain
|
||||
and for the docker/apps state machine to reach a terminal state before
|
||||
restarting. Expect one middlewared restart shortly after every boot; the UI
|
||||
and API are briefly unavailable while it happens, and running services are
|
||||
not affected.
|
||||
|
||||
| Module | What changes | Technique |
|
||||
|---|---|---|
|
||||
| **providers** | `B2RcloneRemote` gains `get_restic_config()` — skipped automatically if TrueNAS already provides one on the class. `restic.py` URL builder is fixed: strips the stray leading slash and converts the slash separator to a colon (`b2:bucket:path`), which is the format restic 0.16.x expects. URL wrapper is a no-op if the URL is already correctly formed. | File patch applied inside the overlayfs upper layer |
|
||||
| **providers** (UI) | The Angular bundle's `filterByProviders` binding is widened from `["STORJ_IX"]` to `["STORJ_IX","S3","B2"]` | In-place text replacement in the compiled JS chunk; original is backed up before patching |
|
||||
| **nested** (opt-in) | `_truecloud_nested.py` is installed into `plugins/cloud/`, and `plugins/cloud/{snapshot,crud}.py` + `plugins/cloud_backup/sync.py` are patched so `snapshot = true` works on a dataset that has child datasets. See [below](#nested-dataset-snapshots). | New module + file patches inside the overlayfs upper layer |
|
||||
|
||||
All changes are **fail-safe**: if a patch cannot be applied (e.g. TrueNAS
|
||||
restructured the relevant code), middlewared starts normally, the affected module
|
||||
is simply inactive, and the reason is logged to `apply.log` in your repo root. The
|
||||
two modules are independent — one failing or going native does not disable the
|
||||
other.
|
||||
|
||||
## Nested-dataset snapshots
|
||||
|
||||
**Opt-in, off by default.** It changes how backups read their source data, so it
|
||||
is never enabled implicitly:
|
||||
|
||||
```bash
|
||||
bash install.sh --enable-nested-snapshots
|
||||
bash install.sh --disable-nested-snapshots
|
||||
```
|
||||
|
||||
With neither flag `install.sh` leaves the setting alone, so `git pull && bash
|
||||
install.sh` won't flip it. The providers module is unaffected either way.
|
||||
|
||||
Validated end to end on a live 252-dataset pool: an unattended scheduled backup
|
||||
of `/mnt/Tap` built a 173-mount staging tree, completed in **18m14s**, and left
|
||||
**zero** orphaned snapshots and **zero** stale mounts behind. The same backup
|
||||
previously stalled at 74% for over 12 hours reading live files.
|
||||
|
||||
Still: verify your own first run actually contains child-dataset data before you
|
||||
rely on it — see [Verifying it works](#verifying-it-works). That advice is not
|
||||
boilerplate; it is the specific thing this feature exists to make true.
|
||||
|
||||
TrueCloud Backup's **Take Snapshot** option makes restic read from a frozen ZFS
|
||||
snapshot instead of live files. Without it the backup reads data *while apps are
|
||||
writing to it* — databases get captured mid-write, and an app that rewrites its
|
||||
files continuously can stall a backup indefinitely as restic chases a moving
|
||||
target.
|
||||
|
||||
Stock TrueNAS refuses to enable it on most real-world paths:
|
||||
|
||||
```
|
||||
[EINVAL] cloud_backup_update.snapshot:
|
||||
This option is only available for datasets that have no further nesting
|
||||
```
|
||||
|
||||
That rules out **any pool running Apps** — every app is its own dataset, usually
|
||||
with `config`/`pgdata` children of its own. On a typical box that is 100+ nested
|
||||
datasets, so the feature is effectively unusable exactly where it matters most.
|
||||
|
||||
### Why stock refuses
|
||||
|
||||
The guard is **correct**. `plugins/cloud/snapshot.py` already takes a *recursive*
|
||||
ZFS snapshot — but it then points restic at the **parent** dataset's
|
||||
`.zfs/snapshot/<snap>/` directory, and ZFS does not expose child datasets
|
||||
through a parent's snapshot directory:
|
||||
|
||||
```
|
||||
/mnt/Tap/.zfs/snapshot/<snap>/apps/ -> 0 entries (children invisible)
|
||||
/mnt/Tap/apps/lidarr/config/.zfs/snapshot/<snap>/ -> the real data
|
||||
```
|
||||
|
||||
So if you just remove the validation, restic walks a near-empty tree, reports
|
||||
SUCCESS, and uploads almost nothing — a green backup job protecting no data. iX
|
||||
gate the config rather than ship a backup that lies about succeeding.
|
||||
|
||||
That is worth spelling out, because deleting those four lines in
|
||||
`plugins/cloud/crud.py` is the obvious "fix" and it is the wrong one. The guard
|
||||
is load-bearing: it has to be *replaced* with a working traversal, not removed.
|
||||
|
||||
### What this patch does instead
|
||||
|
||||
After the (already recursive) snapshot is taken, every descendant dataset's own
|
||||
`.zfs/snapshot/<snap>` is bind-mounted into a **staging tree** that mirrors the
|
||||
original layout, and restic is pointed at the staging root — a complete,
|
||||
consistent, point-in-time view of the whole subtree. Only then is the guard
|
||||
relaxed.
|
||||
|
||||
Safety properties, in order of importance:
|
||||
|
||||
- **Staging failure is loud.** If any descendant cannot be staged, the backup
|
||||
*fails*. A silently-incomplete backup is precisely what the stock guard exists
|
||||
to prevent, and it would be worse than not having the feature at all.
|
||||
- **Post-mount verification** asserts every planned target really is a mountpoint
|
||||
and the staging root is non-empty — so this cannot regress into the empty
|
||||
backup it exists to fix.
|
||||
- **The guard is relaxed last.** `apply.sh` installs the traversal, patches
|
||||
`snapshot.py`, then `sync.py`, and only then `crud.py`. Any partial failure
|
||||
leaves the guard intact and the option merely unavailable — never
|
||||
"guard removed, traversal missing".
|
||||
- Datasets that cannot contribute to a file tree (`mountpoint=none|legacy`,
|
||||
unmounted, locked/encrypted) are skipped and **reported to the log** — never
|
||||
dropped silently.
|
||||
- Scoped to **cloud_backup only**. Cloud Sync (rclone) shares the same
|
||||
validation mixin but has no staging teardown wired in, so its guard is
|
||||
deliberately left in place.
|
||||
|
||||
Side benefit: the staging root is a **stable path per task**, so restic can find
|
||||
its parent snapshot between runs. Stock's `.zfs/snapshot/<name>-<timestamp>/`
|
||||
path changes every run, defeating restic's parent detection and forcing a full
|
||||
re-scan each time.
|
||||
|
||||
### Snapshot lifecycle
|
||||
|
||||
`zfs.snapshot.delete` defaults to **`recursive=False`**, and stock
|
||||
`restic_backup()` calls it with no options. Stock is safe only because its
|
||||
validation means a *recursive* snapshot never actually happens in the field.
|
||||
Enabling nested datasets makes them real: on a 250-dataset pool,
|
||||
`zfs snapshot -r` creates **250 snapshots**, and stock's delete removes only the
|
||||
parent — orphaning **249 on every successful run** (measured, not theorised).
|
||||
|
||||
So the patch owns the whole lifecycle:
|
||||
|
||||
- **Sweeps the parent and every child**, and is idempotent against stock's
|
||||
`finally` winning the race once the mounts are released.
|
||||
- **Records the snapshot in a sidecar file before mounting anything**, so a
|
||||
middlewared restart mid-backup cannot orphan the tree (this patch *schedules*
|
||||
a restart at boot, so that is not hypothetical).
|
||||
- **Reclaims the tree left by a crashed run** instead of overwriting the record.
|
||||
- **Deletes the tree when staging fails** — sync.py's own `finally` deletes
|
||||
*nothing* in that case, because its `snapshot` local never gets assigned.
|
||||
- **Enumerates datasets *after* the snapshot, never before.** A list read
|
||||
beforehand can miss a dataset created in the gap, which the recursive snapshot
|
||||
*would* capture but the staging plan would not — a silent omission.
|
||||
|
||||
**Expected log noise:** stock's delete fails with `EBUSY` while the staging
|
||||
mounts pin the snapshot. You will see one benign `Error deleting snapshot ...`
|
||||
warning per run; the patch then unmounts and deletes the tree for real.
|
||||
|
||||
### Verifying it works
|
||||
|
||||
This feature exists because a backup can report SUCCESS while containing
|
||||
nothing, so check the contents rather than the exit status:
|
||||
|
||||
```bash
|
||||
# 1. Does the restic snapshot actually contain child-dataset data?
|
||||
# Pick a path that lives in a CHILD dataset (e.g. an app's config).
|
||||
midclt call cloud_backup.list_snapshots <task_id> | head
|
||||
|
||||
# 2. List a child-dataset path inside the newest restic snapshot.
|
||||
# If this is empty, the staging tree did not work and you are backing up NOTHING.
|
||||
midclt call cloud_backup.list_snapshot_directory <task_id> "<snapshot_id>" "/apps/lidarr/config"
|
||||
```
|
||||
|
||||
You should see the app's real files (`lidarr.db`, `config.xml`, …). An empty
|
||||
listing means the child datasets were not staged; disable the feature and open an
|
||||
issue.
|
||||
|
||||
```bash
|
||||
# 3. No snapshots may be left behind after a run.
|
||||
zfs list -t snapshot -r <pool> | grep -c cloud_backup- # expect 0 between runs
|
||||
|
||||
# 4. No staging mounts may be left behind.
|
||||
mount | grep truecloud-nested # expect no output
|
||||
```
|
||||
|
||||
### Troubleshooting
|
||||
|
||||
| Symptom | Cause |
|
||||
|---|---|
|
||||
| `This option is only available for datasets that have no further nesting` | Feature not enabled. Run `install.sh --enable-nested-snapshots`, then restart middlewared. |
|
||||
| Backup fails: `dataset '…' has no snapshot '…'; refusing to back up an incomplete tree` | Working as designed — a descendant dataset was not covered by the snapshot. The backup is refused rather than silently omitting that data. |
|
||||
| Backup fails: `snapshot '…' cannot be read (Permission denied)` | The snapshot exists but is unreadable. Middleware runs as root, so this indicates a real permissions problem, not a missing snapshot. |
|
||||
| `cloud_backup-*` snapshots accumulating | The sweep is not running. Check `apply.log` for the nested patch applying, and confirm `sync.py` carries the `TRUECLOUD_PATCH` block. |
|
||||
| Web UI blank after a patch | A bad pattern unbalanced the bundle. `apply.sh` now refuses to write in that case, but if you hit it on an older version: restore `chunk-*.js.pre-truecloud-patch` over the live chunk, then re-run `install.sh`. (`MARKER` makes an already-patched file skip, so the patch cannot heal a corrupted bundle by itself.) |
|
||||
| Stale mounts under `/run/truecloud-nested` | A crashed run. The next backup tears them down. To clear them now: `python3 patch/truecloud_nested.py cleanup` (also run by `uninstall.sh` and `recover.sh`). It names any ZFS snapshot an interrupted run left pinned. |
|
||||
|
||||
## Supported providers after patching
|
||||
|
||||
| Provider | Credential type in TrueNAS |
|
||||
|---|---|
|
||||
| Backblaze B2 (native B2 API) | `B2` |
|
||||
| AWS S3, Wasabi, Cloudflare R2, MinIO, and any S3-compatible endpoint | `S3` |
|
||||
| Storj (unchanged) | `STORJ_IX` |
|
||||
|
||||
## How persistence works
|
||||
|
||||
Two different things must survive two different events:
|
||||
|
||||
| Event | What would be lost | What makes it survive |
|
||||
|---|---|---|
|
||||
| **Reboot** | The overlay holding the patched files lives in `/run` (tmpfs) and vanishes | The PREINIT hook re-runs `apply.sh` on every boot and schedules one middlewared restart to load the result |
|
||||
| **TrueNAS update** | `/usr/` is replaced entirely; custom files in `/etc/` are wiped with the new boot environment | This repo lives on your **data pool**, and the hook registration lives in the **TrueNAS config database** — both survive updates. The first boot after an update is just a normal boot |
|
||||
|
||||
### What happens on every boot
|
||||
|
||||
1. **middlewared starts** with the stock (unpatched) modules. This is
|
||||
unavoidable: PREINIT scripts are executed *by* middlewared
|
||||
(`ix-preinit.service` → `midclt call initshutdownscript.execute_init_tasks`),
|
||||
so nothing registered there can run before it.
|
||||
2. **Pools import** (`ix-zfs.service`), making `/mnt/<pool>` — and this
|
||||
repository — available.
|
||||
3. **`apply.sh` runs** (`ix-preinit.service`): mounts the writable overlay
|
||||
(upper layer in `/run`), patches `b2.py` and `restic.py` on disk inside it,
|
||||
patches the UI bundle, and writes `apply.log` and `hook_status.json`.
|
||||
4. **A deferred restart is scheduled.** The middlewared that is running
|
||||
imported the stock modules in step 1 and never re-imports, so the on-disk
|
||||
patch alone is not enough. `apply.sh` detects it was invoked by middlewared
|
||||
and creates a transient systemd unit (`truecloud-mw-restart`, via
|
||||
`systemd-run --no-block`) running `patch/wait_restart.sh` — detached so it
|
||||
cannot disrupt the remainder of the boot sequence.
|
||||
5. **Once boot has settled, middlewared restarts once** and imports the
|
||||
patched modules from the overlay. `wait_restart.sh` holds the restart until
|
||||
the systemd boot job queue has drained (so in-flight `ix-*` units like
|
||||
`ix-reporting` finish first) *and* middlewared's docker/apps startup has
|
||||
reached a terminal state — plain unit ordering cannot see either, and
|
||||
restarting middlewared while they run kills apps and dashboard reporting
|
||||
for the whole boot. S3/B2 backup support is then active until the next
|
||||
reboot, when the cycle repeats.
|
||||
|
||||
What you will observe: one middlewared restart shortly after every boot (a
|
||||
brief web UI/API blip; running services are unaffected). Between steps 3
|
||||
and 5 there is a short window — typically well under a minute — where the UI
|
||||
already shows S3/B2 (the JS bundle is read from disk per request) but the
|
||||
backend is still stock. A backup job that fires inside that window fails once
|
||||
with `NotImplementedError` and succeeds on its next run; see
|
||||
[Troubleshooting](#troubleshooting) if it persists beyond boot.
|
||||
|
||||
Manual runs of `bash patch/apply.sh` never trigger the restart — that only
|
||||
happens in boot context. `install.sh` and `recover.sh` perform their own
|
||||
explicit restarts instead, which is why a manual re-apply must be followed by
|
||||
`systemctl restart middlewared`.
|
||||
> Storj raised the price of their TrueNAS-integrated tier from **$5/month to
|
||||
> $50/month** in 2026. TrueCloud Backup is the only native TrueNAS feature that gives
|
||||
> you pre-backup ZFS snapshots, restic dedup, scheduled tasks with UI progress, and
|
||||
> dataset-lock integration. Running restic by hand loses all of it. This gets the
|
||||
> feature back with storage you already pay for.
|
||||
|
||||
---
|
||||
|
||||
## Install
|
||||
|
||||
Clone the repository to a **persistent ZFS pool** so it survives OS updates,
|
||||
then run `install.sh` from there:
|
||||
Clone it onto a **pool** (not the boot device — that is wiped on TrueNAS upgrades),
|
||||
then run `install.sh` as root:
|
||||
|
||||
```bash
|
||||
# Replace /mnt/tank with your pool name
|
||||
git clone https://github.com/sudolulo/truenas-truecloud-patch.git \
|
||||
/mnt/tank/truenas-truecloud-patch
|
||||
/mnt/tank/truenas-truecloud-patch # replace `tank` with your pool
|
||||
cd /mnt/tank/truenas-truecloud-patch
|
||||
bash install.sh
|
||||
sudo bash install.sh
|
||||
```
|
||||
|
||||
The directory you clone into becomes the **permanent install location**. The
|
||||
PREINIT boot hook is registered with the exact path you chose, and TrueNAS will
|
||||
call that path on every boot.
|
||||
That registers a PREINIT boot hook, patches middleware in a volatile overlay, and
|
||||
restarts middlewared. **It survives TrueNAS updates** — the patch is re-applied at
|
||||
every boot, never written to the system dataset.
|
||||
|
||||
> **Do not delete or move the repository after install.**
|
||||
> If you need to relocate it, run `bash uninstall.sh` first, move the directory,
|
||||
> then run `bash install.sh` again from the new location. Deleting the repo
|
||||
> without uninstalling leaves a dangling PREINIT hook in the TrueNAS database —
|
||||
> if that happens, see [Emergency recovery](#emergency-recovery) below.
|
||||
Nested-dataset snapshots are **opt-in**:
|
||||
|
||||
Refresh your browser. S3 and B2 credentials now appear in the
|
||||
**Data Protection → TrueCloud Backup → Add** credential dropdown.
|
||||
```bash
|
||||
sudo bash install.sh --enable-nested-snapshots
|
||||
```
|
||||
|
||||
Then create a TrueCloud Backup task in the UI with a B2 or S3 credential, or from
|
||||
[the CLI](docs/cli.md).
|
||||
|
||||
**Check it worked:**
|
||||
|
||||
```bash
|
||||
sudo python3 patch/create_task.py verify
|
||||
```
|
||||
|
||||
If something is wrong, the reason is in `apply.log` — start at
|
||||
[Recovery](docs/recovery.md).
|
||||
|
||||
---
|
||||
|
||||
## TrueNAS compatibility
|
||||
|
||||
<!-- BEGIN COMPAT MATRIX (generated by tools/compat.py --matrix --markdown) -->
|
||||
| TrueNAS | B2/S3 providers | Nested snapshots | Hardware-verified |
|
||||
| --- | --- | --- | --- |
|
||||
| 24.10.2.4 | ok | ok | — |
|
||||
| 25.04.2.6 | ok | ok | — |
|
||||
| 25.10.4 | ok | ok | nested + providers; 252-snapshot recursive backup of /mnt/Tap, 18m |
|
||||
| 26.0.0-BETA.3 _(unreleased)_ | ok | **BROKEN** | — |
|
||||
| master _(unreleased)_ | **BROKEN** | **BROKEN** | — |
|
||||
|
||||
| verdict | meaning |
|
||||
| --- | --- |
|
||||
| **ok** | Every assumption the patch makes about middleware still holds. |
|
||||
| **BROKEN** | middleware changed underneath the patch. `apply.sh` **refuses to apply that module** on this version and leaves TrueNAS stock, so backups keep working — without the module's feature. |
|
||||
| **native** | TrueNAS does this itself now. The module retires; it is not a failure. |
|
||||
|
||||
"ok" means *the patch's assumptions hold*, checked automatically against iX's
|
||||
source. It does not mean a human ran a backup on it — that is the
|
||||
**Hardware-verified** column, which is filled in by hand and only by doing it.
|
||||
<!-- END COMPAT MATRIX -->
|
||||
|
||||
The table is **regenerated daily by CI** against iXsystems' actual middleware source
|
||||
— it is not a claim somebody typed once and forgot.
|
||||
|
||||
**TrueNAS 26: nested snapshots are not supported yet, and upgrading will not break
|
||||
you.** 26 rewrites `cloud_backup` and deletes the ZFS methods this module calls. On
|
||||
26 `apply.sh` finds that the assumptions no longer hold and **does not apply the
|
||||
module**: TrueNAS is left stock, B2/S3 keeps working, nested datasets are simply not
|
||||
covered, and the reason is named in `apply.log`. A broken backup is worse than a
|
||||
missing feature. Details: [How it works](docs/how-it-works.md#truenas-26).
|
||||
|
||||
---
|
||||
|
||||
## Updating
|
||||
|
||||
```bash
|
||||
bash update.sh # to the newest release, with a confirmation
|
||||
bash update.sh --check # show what would happen; change nothing
|
||||
bash update.sh --rollback # undo the last update
|
||||
cd /mnt/tank/truenas-truecloud-patch
|
||||
sudo bash update.sh # newest release
|
||||
sudo bash update.sh --check # what would change?
|
||||
sudo bash update.sh --rollback # back to the previous version
|
||||
```
|
||||
|
||||
It preserves your nested-snapshot opt-in setting, shows you the commits and
|
||||
release notes you don't have yet, and asks before changing anything. It records
|
||||
the previous revision *before* moving, so `--rollback` works even if `install.sh`
|
||||
dies halfway.
|
||||
`update.sh` refuses to run on a dirty checkout, pins you to a release tag, and keeps
|
||||
the previous revision so a rollback is one command. **There is no auto-update**: this
|
||||
patches system internals as root, and a bad commit reaching your box unattended would
|
||||
detonate on the next reboot — v0.0.4 shipped exactly such a bug and took 54 apps
|
||||
down. The manual step *is* the safety gate.
|
||||
|
||||
**Run it by hand. Never from cron or a systemd timer.** This patch injects Python
|
||||
into `middlewared` and re-applies itself at every boot, so an unattended pull would
|
||||
let any bad upstream commit reach your box with no human in the loop and take
|
||||
effect on the next reboot. v0.0.4 shipped exactly such a bug and took every app on
|
||||
the box down. The manual step *is* the safety gate — if you want convenience, watch
|
||||
the [releases](https://github.com/sudolulo/truenas-truecloud-patch/releases) feed,
|
||||
don't automate the pull.
|
||||
---
|
||||
|
||||
It updates to the newest **release tag**, not `main` — `main` can be mid-refactor,
|
||||
and a tag is the tested artifact. `--main` exists if you want unreleased code, and
|
||||
says so loudly.
|
||||
## Update alerts
|
||||
|
||||
## Creating a task via CLI
|
||||
|
||||
If the UI still shows only Storj after refreshing (e.g. the JS bundle pattern
|
||||
changed in a new TrueNAS version), create tasks directly. Run this **on the
|
||||
TrueNAS host** — it talks to the local middleware via `midclt`, so it needs no
|
||||
host address or API key:
|
||||
When a newer release exists, the patch raises a **TrueNAS alert** (the bell in the
|
||||
UI) telling you so. It's on by default and checks once a day.
|
||||
|
||||
```bash
|
||||
# Replace /mnt/tank/truenas-truecloud-patch with your clone path
|
||||
|
||||
# List your cloud credentials to find the right ID
|
||||
python3 /mnt/tank/truenas-truecloud-patch/patch/create_task.py list-credentials
|
||||
|
||||
# Create a task with a B2 credential (id=3).
|
||||
# The restic repo password is read from stdin, so it never lands in your shell
|
||||
# history — nor in any process's argv, where `ps` would expose it.
|
||||
printf '%s' 'restic-repo-password' | \
|
||||
python3 /mnt/tank/truenas-truecloud-patch/patch/create_task.py create \
|
||||
--name "tank-to-b2" \
|
||||
--path /mnt/tank/data \
|
||||
--credential 3 \
|
||||
--bucket my-bucket \
|
||||
--folder backups/tank \
|
||||
--password-stdin \
|
||||
--cache-path /mnt/tank/.restic-cache \
|
||||
--keep-last 14
|
||||
bash install.sh --no-update-alerts # turn it off
|
||||
bash install.sh --update-alerts # turn it back on
|
||||
```
|
||||
|
||||
Omit `--password-stdin` and you'll be prompted for the password instead. `--password
|
||||
<secret>` still works but warns: that password is the encryption key for the whole
|
||||
repository, and a CLI argument persists in your shell history forever.
|
||||
**It will not nag you about a README.** A release whose CHANGELOG contains only a
|
||||
`### Docs` section changed no code, and raises nothing. Anything that touched the
|
||||
system raises an INFO alert; a release with a `### Security` section raises a
|
||||
WARNING. The CHANGELOG's own section headings are the signal, and a security fix
|
||||
anywhere in the range escalates the whole span — a docs-only release on top of a
|
||||
security fix still reports as security.
|
||||
|
||||
> **Always pass `--cache-path`.** Without it TrueNAS runs restic with `--no-cache`,
|
||||
> which re-fetches all repo metadata from the provider every run — glacially slow
|
||||
> on large repos. Point it at a writable dir on a pool with free space.
|
||||
**Release candidates never alert.** They are invisible to `update.sh` and to the
|
||||
alert, which both take the newest plain `vX.Y.Z` tag. That is what lets debugging
|
||||
happen in `-rc` tags instead of in your notification bell — see
|
||||
[Releasing](docs/releasing.md).
|
||||
|
||||
> Versions ≤ 0.1.0 used the `/api/v2.0` REST API with `--host`/`--api-key`; those
|
||||
> flags are now accepted-but-ignored (REST is removed in TrueNAS 26.04).
|
||||
The changelog is read from whichever forge `origin` points at, derived from the
|
||||
remote rather than hard-coded. That is not cosmetic: when the changelog cannot be
|
||||
read, the alert deliberately fires **anyway** rather than risk hiding a security
|
||||
fix — so a wrong URL would not silence the alert, it would make it fire on *every*
|
||||
release, including the documentation-only ones this section promises to suppress.
|
||||
|
||||
### How it works, and why it's built this way
|
||||
|
||||
TrueNAS **cannot raise an alert from the CLI** — `midclt` exposes only
|
||||
`alert.dismiss`, `alert.list`, `alert.list_categories`, `alert.list_policies` and
|
||||
`alert.restore`. Alert *creation* is internal to middlewared, and none of its ~60
|
||||
one-shot alert classes is generic enough to reuse. So the only way to get a real
|
||||
alert is to register an `AlertSource`, which is what `patch/alert_source.py` does.
|
||||
|
||||
That is also the **least invasive** thing this patch does:
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| providers module | **modifies** stock files (appends code to `b2.py`, `restic.py`) |
|
||||
| nested module | **modifies** stock files (3 middleware modules) |
|
||||
| **update alert** | **adds one file. Modifies nothing.** |
|
||||
|
||||
It's the native mechanism — the same one every built-in TrueNAS alert uses — and
|
||||
TrueNAS polls it itself, so there is no cron job and no systemd timer.
|
||||
|
||||
- **Fail-safe.** Every error path returns `None`. It cannot take middlewared down.
|
||||
- **Read-only.** `git ls-remote` plus an HTTPS fetch of the CHANGELOG. It never
|
||||
writes to `.git`, so it cannot leave root-owned objects behind the way a
|
||||
`git fetch` from middlewared (which runs as root) would.
|
||||
- **Removed by `uninstall.sh`.**
|
||||
|
||||
It only *tells* you. It never updates anything — see
|
||||
[Updating](#updating).
|
||||
|
||||
---
|
||||
|
||||
@@ -417,230 +173,39 @@ and restores the original UI bundle from backup.
|
||||
|
||||
---
|
||||
|
||||
## If TrueNAS adds native support
|
||||
## Documentation
|
||||
|
||||
The patch is **two independent modules**, and each retires on its own — TrueNAS
|
||||
is likely to ship one of these natively long before the other, and a module
|
||||
going native must not take the other one down with it.
|
||||
|
||||
| Module | What it does | Detected as native when |
|
||||
|---|---|---|
|
||||
| **providers** | B2/S3 credentials for TrueCloud Backup (`b2.py`, `restic.py`, UI dropdown) | `B2RcloneRemote` carries a real `get_restic_config()` |
|
||||
| **nested** | Snapshots on datasets with child datasets (`plugins/cloud/*`) | the *"no further nesting"* validation is gone from `plugins/cloud/crud.py` |
|
||||
|
||||
At every boot `apply.sh` checks both:
|
||||
|
||||
- **One module goes native** → that module is skipped and logged; the other keeps
|
||||
working, and the patch stays installed.
|
||||
- **Both are done** (native, or nested was never enabled) → the kill switch
|
||||
(`disabled` file) is set, overlays are unmounted, and `apply.log` tells you to
|
||||
run `uninstall.sh`.
|
||||
|
||||
So on a box using only the provider patch, native B2 support retires the whole
|
||||
thing as before. On a box that also uses nested snapshots, native B2 support
|
||||
retires *just* that half.
|
||||
|
||||
Check the log after any TrueNAS update:
|
||||
```bash
|
||||
tail -20 /mnt/tank/truenas-truecloud-patch/apply.log
|
||||
```
|
||||
|
||||
`hook_status.json` reports each module separately (`module.providers`,
|
||||
`module.nested_snapshots`) with an `active` flag and a reason.
|
||||
|
||||
**Scenarios where the auto-detect may not fire** (manual check needed):
|
||||
|
||||
| Scenario | What happens | Action |
|
||||
|---|---|---|
|
||||
| B2 support added to a **base class** (not `B2RcloneRemote` directly) | `__dict__` check misses it; our method shadows native | Uninstall manually |
|
||||
| B2 **credential schema changed** (e.g. `provider["account"]` renamed) | `KeyError` on first backup | Uninstall or update the patch |
|
||||
| **URL builder** fixed but B2 class unchanged | URL wrapper becomes a no-op; no harm, but patch is dead weight | Uninstall at your convenience |
|
||||
|
||||
---
|
||||
|
||||
## After a TrueNAS update
|
||||
|
||||
1. Check the log: `cat /mnt/tank/truenas-truecloud-patch/apply.log | tail -30`
|
||||
2. If you see "WARNING: … pattern not found", the UI patch needs updating.
|
||||
[Open an issue](https://github.com/sudolulo/truenas-truecloud-patch/issues)
|
||||
with your TrueNAS version number.
|
||||
3. The backend patch (B2 support + URL fix) is more stable — check that a
|
||||
B2 backup job still completes successfully after any update.
|
||||
|
||||
---
|
||||
|
||||
## Emergency recovery
|
||||
|
||||
### middlewared won't start
|
||||
|
||||
Run this from the TrueNAS shell (local console, SSH, or the debug shell in
|
||||
the UI):
|
||||
|
||||
```bash
|
||||
bash /mnt/tank/truenas-truecloud-patch/recover.sh
|
||||
```
|
||||
|
||||
Replace the path with your clone location. This creates a kill-switch file
|
||||
(`disabled`) in the repo root, unmounts the overlay so the original files are
|
||||
visible immediately, then restarts middlewared. No reboot required.
|
||||
|
||||
If you cannot run a script and only have a bare shell prompt:
|
||||
|
||||
```bash
|
||||
touch /mnt/tank/truenas-truecloud-patch/disabled
|
||||
systemctl restart middlewared
|
||||
```
|
||||
|
||||
If you don't remember where you cloned the repo (midclt won't work while middlewared is
|
||||
down), find the path two ways:
|
||||
|
||||
```bash
|
||||
# Option 1 — search the filesystem:
|
||||
find /mnt -name "recover.sh" -path "*/truenas-truecloud-patch/*" 2>/dev/null
|
||||
|
||||
# Option 2 — query the TrueNAS database directly:
|
||||
sqlite3 /data/freenas-v1.db \
|
||||
"SELECT script FROM initshutdownscript WHERE comment = 'TrueCloud provider patch (S3/B2)';"
|
||||
```
|
||||
|
||||
The `script` column shows the full path to `patch/apply.sh`; your clone root is one
|
||||
level up (strip `/patch/apply.sh` from the end). Then run the `touch` command above
|
||||
with that path.
|
||||
|
||||
If middlewared **still** won't start after the kill switch is set, the problem
|
||||
is unrelated to this patch. Check:
|
||||
|
||||
```bash
|
||||
journalctl -u middlewared -n 50
|
||||
```
|
||||
|
||||
To re-enable the patch once you have investigated:
|
||||
|
||||
```bash
|
||||
rm /mnt/tank/truenas-truecloud-patch/disabled
|
||||
bash /mnt/tank/truenas-truecloud-patch/patch/apply.sh
|
||||
systemctl restart middlewared # manual apply.sh runs never restart for you
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Web UI is blank or broken
|
||||
|
||||
If the TrueNAS web interface loads blank or shows JavaScript errors, the
|
||||
Angular bundle may have been interrupted mid-write (e.g. power cut during
|
||||
boot). The original bundle is always backed up before patching, so recovery
|
||||
is straightforward:
|
||||
|
||||
```bash
|
||||
# Find the backup (the path varies by TrueNAS version):
|
||||
find /usr/share/truenas /usr/share/truenas-ui /var/www/truenas -name "*.js.pre-truecloud-patch" 2>/dev/null
|
||||
|
||||
# Restore it — substitute the actual path from the find output:
|
||||
mv /usr/share/truenas/webui/main.XXXXXXXX.js.pre-truecloud-patch \
|
||||
/usr/share/truenas/webui/main.XXXXXXXX.js
|
||||
```
|
||||
|
||||
Refresh your browser. The UI will return to normal (Storj-only until the
|
||||
patch re-runs at next reboot, or you run
|
||||
`bash /mnt/tank/truenas-truecloud-patch/patch/apply.sh` manually).
|
||||
|
||||
---
|
||||
|
||||
### Backend verify shows FAIL
|
||||
|
||||
```bash
|
||||
python3 /mnt/tank/truenas-truecloud-patch/patch/create_task.py verify
|
||||
```
|
||||
|
||||
`verify` reports one line per module:
|
||||
|
||||
| Label | Meaning |
|
||||
| | |
|
||||
|---|---|
|
||||
| `[OK ]` | Module is active and applied. |
|
||||
| `[SKIP]` | Module is inactive — either TrueNAS now does it natively, or it is opt-in and switched off. **Not a failure.** `nested_snapshots` shows SKIP on a default install. |
|
||||
| `[FAIL]` | Module is needed but did not apply. |
|
||||
| [Nested-dataset snapshots](docs/nested-snapshots.md) | Why stock refuses, what this does instead, and **how to verify your backups actually contain the data** |
|
||||
| [How it works](docs/how-it-works.md) | What is patched, how it survives updates, the boot sequence, and what happens when TrueNAS goes native |
|
||||
| [Recovery](docs/recovery.md) | middlewared won't start, blank web UI, `verify` shows FAIL |
|
||||
| [CLI](docs/cli.md) | Creating tasks with `create_task.py` |
|
||||
| [Development](docs/releasing.md) | Tests, CI, and the release process |
|
||||
|
||||
If a module shows `[FAIL]`:
|
||||
## What's in the repo
|
||||
|
||||
1. **Check the apply log** for errors during the last boot:
|
||||
```bash
|
||||
tail -40 /mnt/tank/truenas-truecloud-patch/apply.log
|
||||
```
|
||||
2. **Check middlewared's own log** for Python tracebacks:
|
||||
```bash
|
||||
grep -i "truecloud\|traceback\|error" /var/log/middlewared.log 2>/dev/null | tail -30
|
||||
journalctl -u middlewared -n 50
|
||||
```
|
||||
3. **A FAIL is non-fatal.** middlewared runs normally and the other module is
|
||||
unaffected; the failed one is simply inactive. Existing backups are not at
|
||||
risk.
|
||||
4. **If the detail says the module doesn't exist**, a TrueNAS update renamed
|
||||
or restructured the internal API.
|
||||
[Open an issue](https://github.com/sudolulo/truenas-truecloud-patch/issues)
|
||||
with your TrueNAS version number and the full verify output.
|
||||
| Path | What it is |
|
||||
|---|---|
|
||||
| `install.sh` | Register the boot hook, patch, restart middlewared. Also `--enable/--disable-nested-snapshots`. |
|
||||
| `update.sh` | Fetch and apply a newer release. `--check`, `--rollback`, `--to`, `--main`. |
|
||||
| `uninstall.sh` | Remove everything. |
|
||||
| `recover.sh` | Emergency: kill switch + restart against stock files. |
|
||||
| `patch/apply.sh` | The PREINIT script. Runs at **every boot**. |
|
||||
| `patch/truecloud_nested.py` | Nested-dataset staging: plan, mount, verify, tear down, sweep snapshots. |
|
||||
| `patch/create_task.py` | Create tasks with S3/B2 credentials; `verify` the patch state. |
|
||||
| `tools/compat.py` | What the patch assumes about middlewared — and the checker. Run daily by CI *and* at every boot. |
|
||||
| `release.sh` | Cut a release. Two stages, and the second is refused without the first. |
|
||||
|
||||
---
|
||||
## Before you install
|
||||
|
||||
## Troubleshooting
|
||||
- This is **unofficial** and not affiliated with iXsystems.
|
||||
- It patches **internal middleware APIs** with no stability contract. Every patch is
|
||||
fail-safe: if it cannot apply, middlewared starts normally and the reason is logged.
|
||||
- **Test your restores.** True of any backup; more so here. See [Verifying it
|
||||
works](docs/nested-snapshots.md#verifying-it-works).
|
||||
- Filing a TrueNAS bug? **Remove the patch first** and reproduce on a stock system.
|
||||
- Provided as-is, no warranty. See LICENSE.
|
||||
|
||||
**Backups fail with `NotImplementedError` after a reboot**
|
||||
|
||||
The traceback ends in `rclone/base.py` → `raise NotImplementedError` and
|
||||
contains no `_tc_` frames: the running middlewared is executing stock code.
|
||||
Either the deferred restart never fired, or the patch never landed on disk
|
||||
this boot. Diagnose in this order:
|
||||
|
||||
```bash
|
||||
# Did apply.sh run this boot, at which version, and did it schedule the restart?
|
||||
tail -40 /mnt/tank/truenas-truecloud-patch/apply.log
|
||||
|
||||
# Full check — compares the running process against the patch timestamp
|
||||
python3 /mnt/tank/truenas-truecloud-patch/patch/create_task.py verify
|
||||
|
||||
# Did the deferred restart unit run, fail, or never get created?
|
||||
systemctl status truecloud-mw-restart.service
|
||||
journalctl -u truecloud-mw-restart.service --no-pager | tail -20
|
||||
```
|
||||
|
||||
- `verify` reports the process started **before** the patch → the restart
|
||||
didn't happen. `systemctl restart middlewared` fixes it immediately; the
|
||||
journal output above tells you why it was missed.
|
||||
- `apply.log` shows the kill switch is active → `rm .../disabled`, then
|
||||
`bash install.sh`.
|
||||
- `apply.log` has no entry for this boot → the hook didn't run; re-run
|
||||
`bash install.sh` to re-register it.
|
||||
- `apply.log` header shows `[v0.0.3]` or older → update:
|
||||
`git pull && bash install.sh` (v0.0.4 fixed patches not loading after
|
||||
reboot).
|
||||
|
||||
**Apply log** (check after each reboot or install):
|
||||
```bash
|
||||
cat /mnt/tank/truenas-truecloud-patch/apply.log
|
||||
```
|
||||
|
||||
**Verify backend patch is loaded** (while middlewared is running):
|
||||
```bash
|
||||
python3 /mnt/tank/truenas-truecloud-patch/patch/create_task.py verify
|
||||
```
|
||||
Reads `hook_status.json` written by `apply.sh` at boot **and** checks that the
|
||||
running middlewared process started *after* the patches were applied — an
|
||||
on-disk patch that middlewared has not loaded yet is reported as FAIL with
|
||||
instructions. Does not require `--host` or `--api-key`.
|
||||
|
||||
**Middlewared log:**
|
||||
```bash
|
||||
grep truecloud-patch /var/log/middlewared.log 2>/dev/null | tail -20
|
||||
journalctl -u middlewared -n 100 2>/dev/null | grep truecloud-patch
|
||||
```
|
||||
|
||||
**Verify the UI patch** (should print your TrueNAS version):
|
||||
```bash
|
||||
grep -c 'STORJ_IX.*S3.*B2' \
|
||||
$(find /usr/share/truenas -name '*.js' 2>/dev/null) 2>/dev/null \
|
||||
| grep -v ':0'
|
||||
```
|
||||
|
||||
**`create_task.py` — "midclt not found" or permission errors**
|
||||
`create_task.py` now talks to the local middleware via `midclt`, so run it **on
|
||||
the TrueNAS host** (not remotely) as a user with middleware access (root). There
|
||||
is no HTTPS/API-key call anymore, so there is no TLS certificate to configure.
|
||||
Parts of this project were written with AI assistance (Claude); all of it is reviewed
|
||||
and tested before release. Bugs are mine.
|
||||
|
||||
+45
@@ -0,0 +1,45 @@
|
||||
# Creating a task from the CLI
|
||||
|
||||
> Part of [truenas-truecloud-patch](../README.md).
|
||||
|
||||
## Creating a task via CLI
|
||||
|
||||
If the UI still shows only Storj after refreshing (e.g. the JS bundle pattern
|
||||
changed in a new TrueNAS version), create tasks directly. Run this **on the
|
||||
TrueNAS host** — it talks to the local middleware via `midclt`, so it needs no
|
||||
host address or API key:
|
||||
|
||||
```bash
|
||||
# Replace /mnt/tank/truenas-truecloud-patch with your clone path
|
||||
|
||||
# List your cloud credentials to find the right ID
|
||||
python3 /mnt/tank/truenas-truecloud-patch/patch/create_task.py list-credentials
|
||||
|
||||
# Create a task with a B2 credential (id=3).
|
||||
# The restic repo password is read from stdin, so it never lands in your shell
|
||||
# history — nor in any process's argv, where `ps` would expose it.
|
||||
printf '%s' 'restic-repo-password' | \
|
||||
python3 /mnt/tank/truenas-truecloud-patch/patch/create_task.py create \
|
||||
--name "tank-to-b2" \
|
||||
--path /mnt/tank/data \
|
||||
--credential 3 \
|
||||
--bucket my-bucket \
|
||||
--folder backups/tank \
|
||||
--password-stdin \
|
||||
--cache-path /mnt/tank/.restic-cache \
|
||||
--keep-last 14
|
||||
```
|
||||
|
||||
Omit `--password-stdin` and you'll be prompted for the password instead. `--password
|
||||
<secret>` still works but warns: that password is the encryption key for the whole
|
||||
repository, and a CLI argument persists in your shell history forever.
|
||||
|
||||
> **Always pass `--cache-path`.** Without it TrueNAS runs restic with `--no-cache`,
|
||||
> which re-fetches all repo metadata from the provider every run — glacially slow
|
||||
> on large repos. Point it at a writable dir on a pool with free space.
|
||||
|
||||
> Versions ≤ 0.1.0 used the `/api/v2.0` REST API with `--host`/`--api-key`; those
|
||||
> flags are now accepted-but-ignored (REST is removed in TrueNAS 26.04).
|
||||
|
||||
---
|
||||
|
||||
@@ -0,0 +1,209 @@
|
||||
# How it works
|
||||
|
||||
> Part of [truenas-truecloud-patch](../README.md).
|
||||
|
||||
## What is actually patched
|
||||
|
||||
**Nothing in TrueNAS's persistent database or configuration is modified**
|
||||
(other than the boot-hook entry itself). On every boot, `patch/apply.sh` runs
|
||||
as a PREINIT script. It mounts a writable
|
||||
[overlayfs](https://docs.kernel.org/filesystems/overlayfs.html) over the
|
||||
relevant directories in `/usr/` (upper layer in `/run` tmpfs), then patches
|
||||
`b2.py` and `restic.py` inside that overlay. The overlay is volatile — it
|
||||
exists only for the current boot — but the PREINIT script recreates it
|
||||
automatically on every subsequent boot. Nothing in `/usr/` is written to
|
||||
directly.
|
||||
|
||||
PREINIT scripts are executed *by* middlewared, which by then has already
|
||||
imported the stock modules — so after patching, `apply.sh` schedules a single
|
||||
detached middlewared restart (transient systemd unit `truecloud-mw-restart`
|
||||
running `patch/wait_restart.sh`) that loads the patched modules once boot has
|
||||
*actually* settled: the script waits for the systemd boot job queue to drain
|
||||
and for the docker/apps state machine to reach a terminal state before
|
||||
restarting. Expect one middlewared restart shortly after every boot; the UI
|
||||
and API are briefly unavailable while it happens, and running services are
|
||||
not affected.
|
||||
|
||||
| Module | What changes | Technique |
|
||||
|---|---|---|
|
||||
| **providers** | `B2RcloneRemote` gains `get_restic_config()` — skipped automatically if TrueNAS already provides one on the class. `restic.py` URL builder is fixed: strips the stray leading slash and converts the slash separator to a colon (`b2:bucket:path`), which is the format restic 0.16.x expects. URL wrapper is a no-op if the URL is already correctly formed. | File patch applied inside the overlayfs upper layer |
|
||||
| **providers** (UI) | The Angular bundle's `filterByProviders` binding is widened from `["STORJ_IX"]` to `["STORJ_IX","S3","B2"]` | In-place text replacement in the compiled JS chunk; original is backed up before patching |
|
||||
| **nested** (opt-in) | `_truecloud_nested.py` is installed into `plugins/cloud/`, and `plugins/cloud/{snapshot,crud}.py` + `plugins/cloud_backup/sync.py` are patched so `snapshot = true` works on a dataset that has child datasets. See [below](nested-snapshots.md). | New module + file patches inside the overlayfs upper layer |
|
||||
|
||||
All changes are **fail-safe**: if a patch cannot be applied (e.g. TrueNAS
|
||||
restructured the relevant code), middlewared starts normally, the affected module
|
||||
is simply inactive, and the reason is logged to `apply.log` in your repo root. The
|
||||
two modules are independent — one failing or going native does not disable the
|
||||
other.
|
||||
|
||||
|
||||
## How persistence works
|
||||
|
||||
Two different things must survive two different events:
|
||||
|
||||
| Event | What would be lost | What makes it survive |
|
||||
|---|---|---|
|
||||
| **Reboot** | The overlay holding the patched files lives in `/run` (tmpfs) and vanishes | The PREINIT hook re-runs `apply.sh` on every boot and schedules one middlewared restart to load the result |
|
||||
| **TrueNAS update** | `/usr/` is replaced entirely; custom files in `/etc/` are wiped with the new boot environment | This repo lives on your **data pool**, and the hook registration lives in the **TrueNAS config database** — both survive updates. The first boot after an update is just a normal boot |
|
||||
|
||||
### What happens on every boot
|
||||
|
||||
1. **middlewared starts** with the stock (unpatched) modules. This is
|
||||
unavoidable: PREINIT scripts are executed *by* middlewared
|
||||
(`ix-preinit.service` → `midclt call initshutdownscript.execute_init_tasks`),
|
||||
so nothing registered there can run before it.
|
||||
2. **Pools import** (`ix-zfs.service`), making `/mnt/<pool>` — and this
|
||||
repository — available.
|
||||
3. **`apply.sh` runs** (`ix-preinit.service`). Before it patches anything it runs
|
||||
the **compatibility preflight** ([`tools/compat.py`](../tools/compat.py)) against
|
||||
the middlewared that is *actually installed*, and **any module whose assumptions
|
||||
no longer hold is not applied** — see [TrueNAS
|
||||
compatibility](../README.md#truenas-compatibility). What survives that check gets applied:
|
||||
it mounts the writable overlay (upper layer in `/run`), patches `b2.py` and
|
||||
`restic.py` on disk inside it, patches the UI bundle, and writes `apply.log` and
|
||||
`hook_status.json`.
|
||||
|
||||
An incompatible module is skipped **for this boot only**. It is not the kill
|
||||
switch: install a release that supports your TrueNAS and the patch re-applies
|
||||
itself on the next boot, with no manual step. (The kill switch is permanent and
|
||||
is set only when TrueNAS has made the patch *unnecessary* — a different
|
||||
situation, and the opposite conclusion.)
|
||||
4. **A deferred restart is scheduled.** The middlewared that is running
|
||||
imported the stock modules in step 1 and never re-imports, so the on-disk
|
||||
patch alone is not enough. `apply.sh` detects it was invoked by middlewared
|
||||
and creates a transient systemd unit (`truecloud-mw-restart`, via
|
||||
`systemd-run --no-block`) running `patch/wait_restart.sh` — detached so it
|
||||
cannot disrupt the remainder of the boot sequence.
|
||||
5. **Once boot has settled, middlewared restarts once** and imports the
|
||||
patched modules from the overlay. `wait_restart.sh` holds the restart until
|
||||
the systemd boot job queue has drained (so in-flight `ix-*` units like
|
||||
`ix-reporting` finish first) *and* middlewared's docker/apps startup has
|
||||
reached a terminal state — plain unit ordering cannot see either, and
|
||||
restarting middlewared while they run kills apps and dashboard reporting
|
||||
for the whole boot. S3/B2 backup support is then active until the next
|
||||
reboot, when the cycle repeats.
|
||||
|
||||
What you will observe: one middlewared restart shortly after every boot (a
|
||||
brief web UI/API blip; running services are unaffected). Between steps 3
|
||||
and 5 there is a short window — typically well under a minute — where the UI
|
||||
already shows S3/B2 (the JS bundle is read from disk per request) but the
|
||||
backend is still stock. A backup job that fires inside that window fails once
|
||||
with `NotImplementedError` and succeeds on its next run; see
|
||||
[Troubleshooting](recovery.md) if it persists beyond boot.
|
||||
|
||||
Manual runs of `bash patch/apply.sh` never trigger the restart — that only
|
||||
happens in boot context. `install.sh` and `recover.sh` perform their own
|
||||
explicit restarts instead, which is why a manual re-apply must be followed by
|
||||
`systemctl restart middlewared`.
|
||||
|
||||
---
|
||||
|
||||
|
||||
## If TrueNAS adds native support
|
||||
|
||||
The patch is **two independent modules**, and each retires on its own — TrueNAS
|
||||
is likely to ship one of these natively long before the other, and a module
|
||||
going native must not take the other one down with it.
|
||||
|
||||
| Module | What it does | Detected as native when |
|
||||
|---|---|---|
|
||||
| **providers** | B2/S3 credentials for TrueCloud Backup (`b2.py`, `restic.py`, UI dropdown) | `B2RcloneRemote` carries a real `get_restic_config()` |
|
||||
| **nested** | Snapshots on datasets with child datasets (`plugins/cloud/*`) | the *"no further nesting"* validation is gone from `plugins/cloud/crud.py` |
|
||||
|
||||
At every boot `apply.sh` checks both:
|
||||
|
||||
- **One module goes native** → that module is skipped and logged; the other keeps
|
||||
working, and the patch stays installed.
|
||||
- **Both are done** (native, or nested was never enabled) → the kill switch
|
||||
(`disabled` file) is set, overlays are unmounted, and `apply.log` tells you to
|
||||
run `uninstall.sh`.
|
||||
|
||||
So on a box using only the provider patch, native B2 support retires the whole
|
||||
thing as before. On a box that also uses nested snapshots, native B2 support
|
||||
retires *just* that half.
|
||||
|
||||
Check the log after any TrueNAS update:
|
||||
```bash
|
||||
tail -20 /mnt/tank/truenas-truecloud-patch/apply.log
|
||||
```
|
||||
|
||||
`hook_status.json` reports each module separately (`module.providers`,
|
||||
`module.nested_snapshots`) with an `active` flag and a reason.
|
||||
|
||||
**Scenarios where the auto-detect may not fire** (manual check needed):
|
||||
|
||||
| Scenario | What happens | Action |
|
||||
|---|---|---|
|
||||
| B2 support added to a **base class** (not `B2RcloneRemote` directly) | `__dict__` check misses it; our method shadows native | Uninstall manually |
|
||||
| B2 **credential schema changed** (e.g. `provider["account"]` renamed) | `KeyError` on first backup | Uninstall or update the patch |
|
||||
| **URL builder** fixed but B2 class unchanged | URL wrapper becomes a no-op; no harm, but patch is dead weight | Uninstall at your convenience |
|
||||
|
||||
---
|
||||
|
||||
## After a TrueNAS update
|
||||
|
||||
A TrueNAS update replaces `/usr/` wholesale, wiping the patch. You do **not** need
|
||||
to reinstall: `patch/apply.sh` runs at every boot and re-applies itself from your
|
||||
clone. But it targets internal APIs with no stability contract, so an update *can*
|
||||
break it — and the failure is quiet by design (middlewared starts fine; the patch
|
||||
just doesn't).
|
||||
|
||||
**Check the log after any TrueNAS update:**
|
||||
|
||||
```bash
|
||||
tail -30 /mnt/tank/truenas-truecloud-patch/apply.log
|
||||
python3 /mnt/tank/truenas-truecloud-patch/patch/create_task.py verify
|
||||
```
|
||||
|
||||
| What you see | What it means |
|
||||
|---|---|
|
||||
| `[OK] providers`, `[OK]`/`[SKIP] nested_snapshots` | Fine. Nothing to do. |
|
||||
| `WARNING: … pattern not found` (UI) | The Angular bundle changed. The UI dropdown reverts to Storj-only, but **backups keep working** — create tasks with `create_task.py` meanwhile, and [open an issue](https://github.com/sudolulo/truenas-truecloud-patch/issues) with your TrueNAS version. |
|
||||
| `WARNING: truecloud-patch is NOT COMPATIBLE with this TrueNAS version` | This TrueNAS changed middleware underneath the patch, and the named module was **deliberately not applied** — see `incompatible.json` for exactly which assumption broke. TrueNAS is left stock, so nothing is half-patched. Check [TrueNAS compatibility](../README.md#truenas-compatibility), then `bash update.sh` once a release supports your version; it re-applies itself on the next boot. This is **not** the kill switch and needs no manual reset. |
|
||||
| `[FAIL] providers` | **Your B2/S3 backups will not run.** middlewared is fine, but the credential/URL handling is gone. Open an issue with your version. |
|
||||
| `[FAIL] nested_snapshots` | The stock guard is back, so tasks with `snapshot = true` on a nested dataset will fail validation. Turn the option off on those tasks until it's fixed. |
|
||||
|
||||
"Fail-safe" means *the box stays up* — not that your backups keep running. A
|
||||
`[FAIL] providers` is a broken backup, so check the log rather than assume.
|
||||
|
||||
Then update the patch itself if a newer release fixes it:
|
||||
|
||||
```bash
|
||||
bash /mnt/tank/truenas-truecloud-patch/update.sh
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
|
||||
|
||||
## TrueNAS 26
|
||||
|
||||
TrueNAS 26 changes three things underneath the nested module. **Each one alone is
|
||||
backup-breaking**, and none of them is visible from the `cloud_backup` files:
|
||||
|
||||
| what changed | what it would do |
|
||||
| --- | --- |
|
||||
| `cloud_backup` rewritten **async → synchronous** | an `async def` wrapper hands `sync.py` a coroutine where it unpacks a tuple |
|
||||
| `get_dataset_recursive()` **deleted** from `plugins/cloud/snapshot.py` | `NameError` — the injected block called it out of the host module's namespace |
|
||||
| `plugins/zfs_/dataset.py` and `zfs_/snapshot.py` **deleted** | `zfs.dataset.query`, `zfs.snapshot.query` and `zfs.snapshot.delete` all vanish. 26 uses `filesystem.statfs` and `zfs.resource.*` |
|
||||
|
||||
The first two are fixed: the patch reads which flavour of `cloud_backup` your box
|
||||
declares and injects the wrapper that matches (one implementation of the real logic,
|
||||
two thin wrappers), and it carries its own copy of the deleted helper.
|
||||
|
||||
The third is **not** fixed, and is why 26 reports BROKEN. Porting it means rewriting
|
||||
the module's ZFS calls onto 26's new API, and no single API spans 24.10 through 26 —
|
||||
so it needs a real 26 box to verify against, not a plausible-looking diff. Shipping a
|
||||
port nobody has run is exactly the failure this project exists to avoid.
|
||||
|
||||
It is also the row that would have hurt most. `zfs.snapshot.delete` is what sweeps the
|
||||
recursive snapshot; without it, **every run would orphan one snapshot per descendant
|
||||
dataset — 250 on a real pool — forever.** The compatibility check caught it only
|
||||
because it now asserts the middleware *methods the patch calls*, not just the symbols
|
||||
it wraps.
|
||||
|
||||
`master` (development after 26) reports BROKEN too: iXsystems are still reshaping
|
||||
these functions there, renaming `middleware` → `context` and `cloud_backup` → `entry`
|
||||
and adding a required `credentials` parameter. That is a moving target and is
|
||||
deliberately not chased; the check keeps reporting it until it settles into a beta,
|
||||
which is when it becomes worth fixing.
|
||||
@@ -0,0 +1,194 @@
|
||||
# Nested-dataset snapshots
|
||||
|
||||
> Part of [truenas-truecloud-patch](../README.md).
|
||||
|
||||
## Nested-dataset snapshots
|
||||
|
||||
**Opt-in, off by default.** It changes how backups read their source data, so it
|
||||
is never enabled implicitly:
|
||||
|
||||
```bash
|
||||
bash install.sh --enable-nested-snapshots
|
||||
bash install.sh --disable-nested-snapshots
|
||||
```
|
||||
|
||||
With neither flag `install.sh` leaves the setting alone, so `git pull && bash
|
||||
install.sh` won't flip it. The providers module is unaffected either way.
|
||||
|
||||
Validated end to end on a live 252-dataset pool: an unattended scheduled backup
|
||||
of `/mnt/Tap` built a 173-mount staging tree, completed in **18m14s**, and left
|
||||
**zero** orphaned snapshots and **zero** stale mounts behind. The same backup
|
||||
previously stalled at 74% for over 12 hours reading live files.
|
||||
|
||||
Still: verify your own first run actually contains child-dataset data before you
|
||||
rely on it — see [Verifying it works](#verifying-it-works). That advice is not
|
||||
boilerplate; it is the specific thing this feature exists to make true.
|
||||
|
||||
TrueCloud Backup's **Take Snapshot** option makes restic read from a frozen ZFS
|
||||
snapshot instead of live files. Without it the backup reads data *while apps are
|
||||
writing to it* — databases get captured mid-write, and an app that rewrites its
|
||||
files continuously can stall a backup indefinitely as restic chases a moving
|
||||
target.
|
||||
|
||||
Stock TrueNAS refuses to enable it on most real-world paths:
|
||||
|
||||
```
|
||||
[EINVAL] cloud_backup_update.snapshot:
|
||||
This option is only available for datasets that have no further nesting
|
||||
```
|
||||
|
||||
That rules out **any pool running Apps** — every app is its own dataset, usually
|
||||
with `config`/`pgdata` children of its own. On a typical box that is 100+ nested
|
||||
datasets, so the feature is effectively unusable exactly where it matters most.
|
||||
|
||||
### Why stock refuses
|
||||
|
||||
The guard is **correct**. `plugins/cloud/snapshot.py` already takes a *recursive*
|
||||
ZFS snapshot — but it then points restic at the **parent** dataset's
|
||||
`.zfs/snapshot/<snap>/` directory, and ZFS does not expose child datasets
|
||||
through a parent's snapshot directory:
|
||||
|
||||
```
|
||||
/mnt/Tap/.zfs/snapshot/<snap>/apps/ -> 0 entries (children invisible)
|
||||
/mnt/Tap/apps/lidarr/config/.zfs/snapshot/<snap>/ -> the real data
|
||||
```
|
||||
|
||||
So if you just remove the validation, restic walks a near-empty tree, reports
|
||||
SUCCESS, and uploads almost nothing — a green backup job protecting no data. iX
|
||||
gate the config rather than ship a backup that lies about succeeding.
|
||||
|
||||
That is worth spelling out, because deleting those four lines in
|
||||
`plugins/cloud/crud.py` is the obvious "fix" and it is the wrong one. The guard
|
||||
is load-bearing: it has to be *replaced* with a working traversal, not removed.
|
||||
|
||||
### What this patch does instead
|
||||
|
||||
After the (already recursive) snapshot is taken, every descendant dataset's own
|
||||
`.zfs/snapshot/<snap>` is bind-mounted into a **staging tree** that mirrors the
|
||||
original layout, and restic is pointed at the staging root — a complete,
|
||||
consistent, point-in-time view of the whole subtree. Only then is the guard
|
||||
relaxed.
|
||||
|
||||
Safety properties, in order of importance:
|
||||
|
||||
- **Staging failure is loud.** If any descendant cannot be staged, the backup
|
||||
*fails*. A silently-incomplete backup is precisely what the stock guard exists
|
||||
to prevent, and it would be worse than not having the feature at all.
|
||||
- **Post-mount verification** asserts every planned target really is a mountpoint
|
||||
and the staging root is non-empty — so this cannot regress into the empty
|
||||
backup it exists to fix.
|
||||
- **The guard is relaxed last.** `apply.sh` installs the traversal, patches
|
||||
`snapshot.py`, then `sync.py`, and only then `crud.py`. Any partial failure
|
||||
leaves the guard intact and the option merely unavailable — never
|
||||
"guard removed, traversal missing".
|
||||
- Datasets that cannot contribute to a file tree (`mountpoint=none|legacy`,
|
||||
unmounted, locked/encrypted) are skipped and **reported to the log** — never
|
||||
dropped silently.
|
||||
- Scoped to **cloud_backup only**. Cloud Sync (rclone) shares the same
|
||||
validation mixin but has no staging teardown wired in, so its guard is
|
||||
deliberately left in place.
|
||||
|
||||
Side benefit: the staging root is a **stable path per task**, so restic can find
|
||||
its parent snapshot between runs. Stock's `.zfs/snapshot/<name>-<timestamp>/`
|
||||
path changes every run, defeating restic's parent detection and forcing a full
|
||||
re-scan each time.
|
||||
|
||||
### Snapshot lifecycle
|
||||
|
||||
> **Two mechanisms clean up, and the second exists because the first can be destroyed.**
|
||||
>
|
||||
> 1. **The sidecar** records exactly which snapshots a run pinned, and is removed only
|
||||
> on a confirmed-clean sweep. Precise, and it survives a middlewared restart.
|
||||
> 2. **The garbage collector** finds leftovers by *name*, so it still works when the
|
||||
> sidecar is gone — and it can be: **the sidecar lives in `/run`, which is tmpfs.** A
|
||||
> reboot mid-backup takes it, and with it the only record of a 250-snapshot tree.
|
||||
>
|
||||
> The collector runs at the start of every backup, after the sidecar reclaim. It will
|
||||
> only touch a snapshot named `<dataset>@<task>-<timestamp>` that is not the current
|
||||
> run's, has **nothing mounted from it** (which is what protects a concurrently-running
|
||||
> backup), and is **over an hour old**. Periodic `auto-*` snapshots, other tasks'
|
||||
> snapshots, and anything you made by hand are structurally out of reach.
|
||||
|
||||
|
||||
> **A snapshot may survive a run, and that is expected.** ZFS **automounts**
|
||||
> `<dataset>/.zfs/snapshot/<snap>` the moment it is read, and holds it for
|
||||
> `zfs_expire_snapshot` seconds (**300** by default) after the last access. So
|
||||
> whatever restic read *last* is still pinned when we try to destroy it, and
|
||||
> `zfs destroy` refuses with `dataset is busy`.
|
||||
>
|
||||
> The patch unmounts those automounts itself and retries, which clears ~255 of 256 on
|
||||
> a real pool. The one that remains is **logged, its sidecar is kept, and the next run
|
||||
> reclaims it before doing anything else** — so the leak is bounded at a single cycle
|
||||
> instead of growing forever. Seeing one `could not delete snapshot … it will be
|
||||
> reclaimed on the next run` in the log is normal. Seeing the count *grow* run over run
|
||||
> is not, and would be a bug.
|
||||
>
|
||||
> This is why the sidecar is removed **only on a confirmed-clean sweep**: it is the
|
||||
> only record those snapshots exist, and a run that dropped it while they were still
|
||||
> around would orphan them permanently. That is precisely what happened before this was
|
||||
> fixed.
|
||||
|
||||
|
||||
`zfs.snapshot.delete` defaults to **`recursive=False`**, and stock
|
||||
`restic_backup()` calls it with no options. Stock is safe only because its
|
||||
validation means a *recursive* snapshot never actually happens in the field.
|
||||
Enabling nested datasets makes them real: on a 250-dataset pool,
|
||||
`zfs snapshot -r` creates **250 snapshots**, and stock's delete removes only the
|
||||
parent — orphaning **249 on every successful run** (measured, not theorised).
|
||||
|
||||
So the patch owns the whole lifecycle:
|
||||
|
||||
- **Sweeps the parent and every child**, and is idempotent against stock's
|
||||
`finally` winning the race once the mounts are released.
|
||||
- **Records the snapshot in a sidecar file before mounting anything**, so a
|
||||
middlewared restart mid-backup cannot orphan the tree (this patch *schedules*
|
||||
a restart at boot, so that is not hypothetical).
|
||||
- **Reclaims the tree left by a crashed run** instead of overwriting the record.
|
||||
- **Deletes the tree when staging fails** — sync.py's own `finally` deletes
|
||||
*nothing* in that case, because its `snapshot` local never gets assigned.
|
||||
- **Enumerates datasets *after* the snapshot, never before.** A list read
|
||||
beforehand can miss a dataset created in the gap, which the recursive snapshot
|
||||
*would* capture but the staging plan would not — a silent omission.
|
||||
|
||||
**Expected log noise:** stock's delete fails with `EBUSY` while the staging
|
||||
mounts pin the snapshot. You will see one benign `Error deleting snapshot ...`
|
||||
warning per run; the patch then unmounts and deletes the tree for real.
|
||||
|
||||
### Verifying it works
|
||||
|
||||
This feature exists because a backup can report SUCCESS while containing
|
||||
nothing, so check the contents rather than the exit status:
|
||||
|
||||
```bash
|
||||
# 1. Does the restic snapshot actually contain child-dataset data?
|
||||
# Pick a path that lives in a CHILD dataset (e.g. an app's config).
|
||||
midclt call cloud_backup.list_snapshots <task_id> | head
|
||||
|
||||
# 2. List a child-dataset path inside the newest restic snapshot.
|
||||
# If this is empty, the staging tree did not work and you are backing up NOTHING.
|
||||
midclt call cloud_backup.list_snapshot_directory <task_id> "<snapshot_id>" "/apps/lidarr/config"
|
||||
```
|
||||
|
||||
You should see the app's real files (`lidarr.db`, `config.xml`, …). An empty
|
||||
listing means the child datasets were not staged; disable the feature and open an
|
||||
issue.
|
||||
|
||||
```bash
|
||||
# 3. No snapshots may be left behind after a run.
|
||||
zfs list -t snapshot -r <pool> | grep -c cloud_backup- # expect 0 between runs
|
||||
|
||||
# 4. No staging mounts may be left behind.
|
||||
mount | grep truecloud-nested # expect no output
|
||||
```
|
||||
|
||||
### Troubleshooting
|
||||
|
||||
| Symptom | Cause |
|
||||
|---|---|
|
||||
| `This option is only available for datasets that have no further nesting` | Feature not enabled. Run `install.sh --enable-nested-snapshots`, then restart middlewared. |
|
||||
| Backup fails: `dataset '…' has no snapshot '…'; refusing to back up an incomplete tree` | Working as designed — a descendant dataset was not covered by the snapshot. The backup is refused rather than silently omitting that data. |
|
||||
| Backup fails: `snapshot '…' cannot be read (Permission denied)` | The snapshot exists but is unreadable. Middleware runs as root, so this indicates a real permissions problem, not a missing snapshot. |
|
||||
| `cloud_backup-*` snapshots accumulating | The sweep is not running. Check `apply.log` for the nested patch applying, and confirm `sync.py` carries the `TRUECLOUD_PATCH` block. |
|
||||
| Web UI blank after a patch | A bad pattern unbalanced the bundle. `apply.sh` now refuses to write in that case, but if you hit it on an older version: restore `chunk-*.js.pre-truecloud-patch` over the live chunk, then re-run `install.sh`. (`MARKER` makes an already-patched file skip, so the patch cannot heal a corrupted bundle by itself.) |
|
||||
| Stale mounts under `/run/truecloud-nested` | A crashed run. The next backup tears them down. To clear them now: `python3 patch/truecloud_nested.py cleanup` (also run by `uninstall.sh` and `recover.sh`). It names any ZFS snapshot an interrupted run left pinned. |
|
||||
|
||||
@@ -0,0 +1,179 @@
|
||||
# Recovery and troubleshooting
|
||||
|
||||
> Part of [truenas-truecloud-patch](../README.md).
|
||||
|
||||
## Emergency recovery
|
||||
|
||||
### middlewared won't start
|
||||
|
||||
Run this from the TrueNAS shell (local console, SSH, or the debug shell in
|
||||
the UI):
|
||||
|
||||
```bash
|
||||
bash /mnt/tank/truenas-truecloud-patch/recover.sh
|
||||
```
|
||||
|
||||
Replace the path with your clone location. This creates a kill-switch file
|
||||
(`disabled`) in the repo root, unmounts the overlay so the original files are
|
||||
visible immediately, then restarts middlewared. No reboot required.
|
||||
|
||||
If you cannot run a script and only have a bare shell prompt:
|
||||
|
||||
```bash
|
||||
touch /mnt/tank/truenas-truecloud-patch/disabled
|
||||
systemctl restart middlewared
|
||||
```
|
||||
|
||||
If you don't remember where you cloned the repo (midclt won't work while middlewared is
|
||||
down), find the path two ways:
|
||||
|
||||
```bash
|
||||
# Option 1 — search the filesystem:
|
||||
find /mnt -name "recover.sh" -path "*/truenas-truecloud-patch/*" 2>/dev/null
|
||||
|
||||
# Option 2 — query the TrueNAS database directly:
|
||||
sqlite3 /data/freenas-v1.db \
|
||||
"SELECT script FROM initshutdownscript WHERE comment = 'TrueCloud provider patch (S3/B2)';"
|
||||
```
|
||||
|
||||
The `script` column shows the full path to `patch/apply.sh`; your clone root is one
|
||||
level up (strip `/patch/apply.sh` from the end). Then run the `touch` command above
|
||||
with that path.
|
||||
|
||||
If middlewared **still** won't start after the kill switch is set, the problem
|
||||
is unrelated to this patch. Check:
|
||||
|
||||
```bash
|
||||
journalctl -u middlewared -n 50
|
||||
```
|
||||
|
||||
To re-enable the patch once you have investigated:
|
||||
|
||||
```bash
|
||||
rm /mnt/tank/truenas-truecloud-patch/disabled
|
||||
bash /mnt/tank/truenas-truecloud-patch/patch/apply.sh
|
||||
systemctl restart middlewared # manual apply.sh runs never restart for you
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Web UI is blank or broken
|
||||
|
||||
If the TrueNAS web interface loads blank or shows JavaScript errors, the
|
||||
Angular bundle may have been interrupted mid-write (e.g. power cut during
|
||||
boot). The original bundle is always backed up before patching, so recovery
|
||||
is straightforward:
|
||||
|
||||
```bash
|
||||
# Find the backup (the path varies by TrueNAS version):
|
||||
find /usr/share/truenas /usr/share/truenas-ui /var/www/truenas -name "*.js.pre-truecloud-patch" 2>/dev/null
|
||||
|
||||
# Restore it — substitute the actual path from the find output:
|
||||
mv /usr/share/truenas/webui/main.XXXXXXXX.js.pre-truecloud-patch \
|
||||
/usr/share/truenas/webui/main.XXXXXXXX.js
|
||||
```
|
||||
|
||||
Refresh your browser. The UI will return to normal (Storj-only until the
|
||||
patch re-runs at next reboot, or you run
|
||||
`bash /mnt/tank/truenas-truecloud-patch/patch/apply.sh` manually).
|
||||
|
||||
---
|
||||
|
||||
### Backend verify shows FAIL
|
||||
|
||||
```bash
|
||||
python3 /mnt/tank/truenas-truecloud-patch/patch/create_task.py verify
|
||||
```
|
||||
|
||||
`verify` reports one line per module:
|
||||
|
||||
| Label | Meaning |
|
||||
|---|---|
|
||||
| `[OK ]` | Module is active and applied. |
|
||||
| `[SKIP]` | Module is inactive — either TrueNAS now does it natively, or it is opt-in and switched off. **Not a failure.** `nested_snapshots` shows SKIP on a default install. |
|
||||
| `[FAIL]` | Module is needed but did not apply. |
|
||||
|
||||
If a module shows `[FAIL]`:
|
||||
|
||||
1. **Check the apply log** for errors during the last boot:
|
||||
```bash
|
||||
tail -40 /mnt/tank/truenas-truecloud-patch/apply.log
|
||||
```
|
||||
2. **Check middlewared's own log** for Python tracebacks:
|
||||
```bash
|
||||
grep -i "truecloud\|traceback\|error" /var/log/middlewared.log 2>/dev/null | tail -30
|
||||
journalctl -u middlewared -n 50
|
||||
```
|
||||
3. **A FAIL is non-fatal.** middlewared runs normally and the other module is
|
||||
unaffected; the failed one is simply inactive. Existing backups are not at
|
||||
risk.
|
||||
4. **If the detail says the module doesn't exist**, a TrueNAS update renamed
|
||||
or restructured the internal API.
|
||||
[Open an issue](https://github.com/sudolulo/truenas-truecloud-patch/issues)
|
||||
with your TrueNAS version number and the full verify output.
|
||||
|
||||
---
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
**Backups fail with `NotImplementedError` after a reboot**
|
||||
|
||||
The traceback ends in `rclone/base.py` → `raise NotImplementedError` and
|
||||
contains no `_tc_` frames: the running middlewared is executing stock code.
|
||||
Either the deferred restart never fired, or the patch never landed on disk
|
||||
this boot. Diagnose in this order:
|
||||
|
||||
```bash
|
||||
# Did apply.sh run this boot, at which version, and did it schedule the restart?
|
||||
tail -40 /mnt/tank/truenas-truecloud-patch/apply.log
|
||||
|
||||
# Full check — compares the running process against the patch timestamp
|
||||
python3 /mnt/tank/truenas-truecloud-patch/patch/create_task.py verify
|
||||
|
||||
# Did the deferred restart unit run, fail, or never get created?
|
||||
systemctl status truecloud-mw-restart.service
|
||||
journalctl -u truecloud-mw-restart.service --no-pager | tail -20
|
||||
```
|
||||
|
||||
- `verify` reports the process started **before** the patch → the restart
|
||||
didn't happen. `systemctl restart middlewared` fixes it immediately; the
|
||||
journal output above tells you why it was missed.
|
||||
- `apply.log` shows the kill switch is active → `rm .../disabled`, then
|
||||
`bash install.sh`.
|
||||
- `apply.log` has no entry for this boot → the hook didn't run; re-run
|
||||
`bash install.sh` to re-register it.
|
||||
- `apply.log` header shows `[v0.0.3]` or older → update:
|
||||
`git pull && bash install.sh` (v0.0.4 fixed patches not loading after
|
||||
reboot).
|
||||
|
||||
**Apply log** (check after each reboot or install):
|
||||
```bash
|
||||
cat /mnt/tank/truenas-truecloud-patch/apply.log
|
||||
```
|
||||
|
||||
**Verify backend patch is loaded** (while middlewared is running):
|
||||
```bash
|
||||
python3 /mnt/tank/truenas-truecloud-patch/patch/create_task.py verify
|
||||
```
|
||||
Reads `hook_status.json` written by `apply.sh` at boot **and** checks that the
|
||||
running middlewared process started *after* the patches were applied — an
|
||||
on-disk patch that middlewared has not loaded yet is reported as FAIL with
|
||||
instructions. Does not require `--host` or `--api-key`.
|
||||
|
||||
**Middlewared log:**
|
||||
```bash
|
||||
grep truecloud-patch /var/log/middlewared.log 2>/dev/null | tail -20
|
||||
journalctl -u middlewared -n 100 2>/dev/null | grep truecloud-patch
|
||||
```
|
||||
|
||||
**Verify the UI patch** (should print your TrueNAS version):
|
||||
```bash
|
||||
grep -c 'STORJ_IX.*S3.*B2' \
|
||||
$(find /usr/share/truenas -name '*.js' 2>/dev/null) 2>/dev/null \
|
||||
| grep -v ':0'
|
||||
```
|
||||
|
||||
**`create_task.py` — "midclt not found" or permission errors**
|
||||
`create_task.py` now talks to the local middleware via `midclt`, so run it **on
|
||||
the TrueNAS host** (not remotely) as a user with middleware access (root). There
|
||||
is no HTTPS/API-key call anymore, so there is no TLS certificate to configure.
|
||||
@@ -0,0 +1,98 @@
|
||||
# Development and releasing
|
||||
|
||||
> Part of [truenas-truecloud-patch](../README.md).
|
||||
|
||||
## Development
|
||||
|
||||
Parts of this project were written with AI assistance (Claude). All of it is
|
||||
reviewed and tested before release; the test suite and CI exist in large part to
|
||||
make that review meaningful. Bugs are mine.
|
||||
|
||||
```bash
|
||||
pip install pytest ruff
|
||||
ruff check patch tests tools
|
||||
pytest tests
|
||||
```
|
||||
|
||||
CI runs shellcheck, `bash -n`, ruff, and pytest on Python 3.11–3.13. Two checks
|
||||
are worth calling out, because nothing else would catch what they catch:
|
||||
|
||||
- The tests **`compile()` the `*_BLOCK` strings** in `patch/apply.sh`. Those are
|
||||
Python source appended into live `middlewared` modules — a syntax error there
|
||||
breaks the box at boot, and they're string literals, so nothing else type-checks
|
||||
them.
|
||||
- CI asserts **every script declares the same version**, and that it matches the
|
||||
newest CHANGELOG entry. `VERSION=` had silently drifted to three different
|
||||
values across the scripts before anything checked.
|
||||
|
||||
The project is hosted on **Gitea** (`git.onetick.ninja/flan/truenas-truecloud-patch`)
|
||||
and mirrored to GitHub. Both run the same workflows — Gitea reads
|
||||
`.github/workflows/` too — so a change is checked twice, on two independent runners.
|
||||
|
||||
---
|
||||
|
||||
## Releasing
|
||||
|
||||
**Every release interrupts every user.** An update alert fires on each installed
|
||||
box (see [Update alerts](../README.md#update-alerts)), so a release that exists only to fix the
|
||||
last release teaches people to dismiss the alert — and one day that alert will be
|
||||
carrying a security fix. This project cut twelve releases in a single day once.
|
||||
Never again, and not by good intentions: by a gate.
|
||||
|
||||
### The rule
|
||||
|
||||
> A stable `vX.Y.Z` may only be published if a `vX.Y.Z-rcN` tag points at the
|
||||
> **same commit**.
|
||||
|
||||
Release candidates are **invisible to users**: `update.sh` and the update alert both
|
||||
take the newest plain `vX.Y.Z` tag, so an `-rc` is never offered as an update. All
|
||||
the debugging therefore happens across `rc1`, `rc2`, `rc3` — at nobody's expense —
|
||||
instead of across `v0.5.0`, `v0.5.1`, `v0.5.2`, at everybody's.
|
||||
|
||||
"The candidate passed, then I pushed one more little fix" is refused **by name**.
|
||||
That is not hypothetical; it is exactly how v0.5.1 happened.
|
||||
|
||||
### Day to day
|
||||
|
||||
You don't touch the release machinery. Write your changes under `## Unreleased` in
|
||||
`CHANGELOG.md` and push to `main`. `main` is a work surface — it is allowed to be
|
||||
mid-thought. Releasing is a separate, deliberate act.
|
||||
|
||||
### Cutting a release
|
||||
|
||||
```bash
|
||||
bash release.sh 0.6.0 --check # what would ship? what is the next rc?
|
||||
|
||||
bash release.sh 0.6.0 --rc # promotes `## Unreleased` -> v0.6.0, stamps every
|
||||
# VERSION=, tags v0.6.0-rc1, pushes. Users see nothing.
|
||||
|
||||
# ... install it on a real box. Exercise it. Break it. ...
|
||||
# Found a bug? Fix it on main, then `bash release.sh 0.6.0 --rc` again -> rc2.
|
||||
|
||||
bash release.sh 0.6.0 --promote # publishes v0.6.0. REFUSED unless an rc points here.
|
||||
```
|
||||
|
||||
### The gates, and where they live
|
||||
|
||||
The logic is Python so it can be unit-tested; CI is the enforcement boundary
|
||||
because it is the only actor holding the token that publishes. `release.sh` runs the
|
||||
**same** code locally so you fail in 200 ms instead of after a push.
|
||||
|
||||
| gate | enforces | where |
|
||||
| --- | --- | --- |
|
||||
| [`release_notes.py check`](../tools/release_notes.py) | every script's `VERSION=` matches the tag; the CHANGELOG section exists and is non-empty; **nothing is stranded under `## Unreleased`** | `release.sh` + CI |
|
||||
| [`release_gate.py`](../tools/release_gate.py) | **an rc points at this exact commit** | `release.sh` + CI |
|
||||
| the full suite | ruff, pytest, shellcheck, `bash -n` — re-run against the *tagged* commit | CI |
|
||||
|
||||
A release's body **is** its `CHANGELOG.md` section — there is no second place to
|
||||
write release notes, and therefore no second place for them to go stale. Releases
|
||||
are published on both forges.
|
||||
|
||||
### Why the alert doesn't nag
|
||||
|
||||
A release whose CHANGELOG contains only a `### Docs` section changed no code, and
|
||||
raises **no alert**. Candidates raise no alert either. So the only thing that ever
|
||||
interrupts a user is a real, complete change — which is the entire point.
|
||||
|
||||
---
|
||||
|
||||
+81
-3
@@ -18,7 +18,7 @@
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
VERSION="0.4.0"
|
||||
VERSION="0.6.1"
|
||||
|
||||
# The directory containing install.sh is the permanent install location.
|
||||
PATCH_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
@@ -31,6 +31,8 @@ _NESTED_MARKER="$PATCH_DIR/nested_snapshots_enabled"
|
||||
# `git pull`) must never flip it on or off by itself: with neither flag given,
|
||||
# whatever was chosen previously is preserved.
|
||||
_nested_choice=""
|
||||
_alert_choice=""
|
||||
_ALERT_MARKER="$PATCH_DIR/update_alerts_disabled"
|
||||
|
||||
usage() {
|
||||
cat <<USAGE
|
||||
@@ -42,6 +44,8 @@ Options:
|
||||
Stock TrueNAS refuses this; see README. Off by
|
||||
default because it changes how backups read data.
|
||||
--disable-nested-snapshots Turn it back off; the stock guard is restored.
|
||||
--no-update-alerts Do not raise a TrueNAS alert when an update exists.
|
||||
--update-alerts Re-enable those alerts (they are on by default).
|
||||
-h, --help Show this help.
|
||||
|
||||
With neither flag, the current setting is left unchanged.
|
||||
@@ -52,6 +56,8 @@ while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--enable-nested-snapshots) _nested_choice="on" ;;
|
||||
--disable-nested-snapshots) _nested_choice="off" ;;
|
||||
--no-update-alerts) _alert_choice="off" ;;
|
||||
--update-alerts) _alert_choice="on" ;;
|
||||
-h|--help) usage; exit 0 ;;
|
||||
*)
|
||||
echo "ERROR: unknown option: $1" >&2
|
||||
@@ -82,6 +88,44 @@ if [ "$(id -u)" -ne 0 ]; then
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# ── Minimum TrueNAS version ───────────────────────────────────────────────────
|
||||
#
|
||||
# TrueCloud Backup -- the feature this whole project extends -- was introduced in
|
||||
# 24.10. On anything older, `plugins/cloud_backup/` does not exist at all: there is
|
||||
# no restic, no cloud_backup task type, and nothing for the patch to attach to. It
|
||||
# would not break the box, it would simply do nothing, silently, while the user
|
||||
# believed their backups were configured. Say no clearly instead.
|
||||
#
|
||||
# A version we cannot PARSE is not a version we may refuse on: warn and continue.
|
||||
# Refusing to install over a string we failed to read would be a worse failure than
|
||||
# the one being prevented.
|
||||
MIN_TRUENAS="24.10"
|
||||
|
||||
_tc_version_raw=""
|
||||
if command -v midclt &>/dev/null; then
|
||||
_tc_version_raw=$(midclt call system.version 2>/dev/null || true) # TrueNAS-25.10.4
|
||||
fi
|
||||
# Both spellings are real: modern releases report `TrueNAS-25.10.4`, older ones
|
||||
# `TrueNAS-SCALE-24.04.2` -- and the SCALE- form is used by exactly the versions
|
||||
# this gate exists to turn away, so failing to parse it would let them through.
|
||||
_tc_version=$(printf '%s' "$_tc_version_raw" \
|
||||
| sed -n 's/^TrueNAS-\(SCALE-\)\{0,1\}\([0-9]\{1,\}\.[0-9]\{1,\}\).*/\2/p')
|
||||
|
||||
if [ -z "$_tc_version" ]; then
|
||||
echo "WARNING: could not determine the TrueNAS version" \
|
||||
"${_tc_version_raw:+(got '${_tc_version_raw}')}."
|
||||
echo "WARNING: this patch requires TrueNAS SCALE ${MIN_TRUENAS} or newer. Continuing anyway."
|
||||
echo ""
|
||||
elif [ "$(printf '%s\n%s\n' "$MIN_TRUENAS" "$_tc_version" | sort -V | head -1)" != "$MIN_TRUENAS" ]; then
|
||||
echo "ERROR: TrueNAS ${_tc_version} is too old — this patch requires ${MIN_TRUENAS} or newer." >&2
|
||||
echo "" >&2
|
||||
echo " TrueCloud Backup does not exist before ${MIN_TRUENAS}, so there is nothing" >&2
|
||||
echo " here for the patch to extend. Upgrade TrueNAS first." >&2
|
||||
exit 1
|
||||
else
|
||||
echo "TrueNAS ${_tc_version} (minimum ${MIN_TRUENAS}) — ok"
|
||||
fi
|
||||
|
||||
if ! command -v midclt &>/dev/null; then
|
||||
echo "ERROR: midclt not found. Run this script on TrueNAS SCALE." >&2
|
||||
exit 1
|
||||
@@ -95,8 +139,14 @@ fi
|
||||
# ── Set permissions ───────────────────────────────────────────────────────────
|
||||
|
||||
echo "Setting permissions ..."
|
||||
chmod +x "$PATCH_DIR/patch/apply.sh" "$PATCH_DIR/patch/create_task.py" \
|
||||
"$PATCH_DIR/recover.sh" "$PATCH_DIR/uninstall.sh" "$PATCH_DIR/update.sh"
|
||||
# Guard each path: under `set -e` a chmod on a missing file aborts the install.
|
||||
# The file set changes between versions, so `update.sh --rollback` to an older
|
||||
# revision must not be killed by a name this version happens to know about.
|
||||
for _exe in patch/apply.sh patch/create_task.py recover.sh uninstall.sh update.sh; do
|
||||
if [ -f "$PATCH_DIR/$_exe" ]; then
|
||||
chmod +x "$PATCH_DIR/$_exe"
|
||||
fi
|
||||
done
|
||||
echo "Done."
|
||||
echo ""
|
||||
|
||||
@@ -186,6 +236,34 @@ case "$_nested_choice" in
|
||||
esac
|
||||
echo ""
|
||||
|
||||
# ── Update alerts (on by default) ─────────────────────────────────────────────
|
||||
# TrueNAS cannot raise an alert from the CLI (midclt exposes only dismiss/list/
|
||||
# restore), so this installs an AlertSource into middlewared/alert/source/ — the
|
||||
# same mechanism every built-in TrueNAS alert uses. It ADDS a file and modifies
|
||||
# none, which makes it the least invasive thing this patch does.
|
||||
#
|
||||
# It only alerts for releases that changed something: a documentation-only release
|
||||
# raises nothing.
|
||||
|
||||
case "$_alert_choice" in
|
||||
off)
|
||||
touch "$_ALERT_MARKER"
|
||||
echo "Update alerts: DISABLED (apply.sh will remove the alert source)."
|
||||
;;
|
||||
on)
|
||||
rm -f "$_ALERT_MARKER"
|
||||
echo "Update alerts: enabled."
|
||||
;;
|
||||
*)
|
||||
if [ -f "$_ALERT_MARKER" ]; then
|
||||
echo "Update alerts: disabled (unchanged)."
|
||||
else
|
||||
echo "Update alerts: enabled (checks daily; docs-only releases are ignored)."
|
||||
fi
|
||||
;;
|
||||
esac
|
||||
echo ""
|
||||
|
||||
# ── Apply now ─────────────────────────────────────────────────────────────────
|
||||
|
||||
echo "Applying patches ..."
|
||||
|
||||
@@ -0,0 +1,258 @@
|
||||
"""TrueNAS alert: a truecloud-patch update is available.
|
||||
|
||||
Installed by patch/apply.sh into middlewared/alert/source/, where middlewared
|
||||
discovers and polls it natively — no cron job, no systemd timer.
|
||||
|
||||
@PATCH_DIR@ is substituted at install time.
|
||||
|
||||
Two rules govern this file:
|
||||
|
||||
1. **It must never break middlewared.** It runs inside the alert framework on a
|
||||
timer. Every failure path returns None (no alert) rather than raising.
|
||||
|
||||
2. **It must not nag.** A release whose CHANGELOG only has a "### Docs" section
|
||||
changed no code, and nobody wants an alert because a README was reworded. The
|
||||
CHANGELOG's own section headings are the signal — see tools/release_notes.py.
|
||||
|
||||
It also never writes to the repository. `git ls-remote` is read-only and the
|
||||
CHANGELOG is fetched over HTTPS, so this cannot leave root-owned objects in .git
|
||||
the way a `git fetch` from middlewared (running as root) would.
|
||||
"""
|
||||
|
||||
import datetime
|
||||
import importlib.util
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import urllib.request
|
||||
|
||||
from middlewared.alert.base import (
|
||||
Alert,
|
||||
AlertCategory,
|
||||
AlertClass,
|
||||
AlertLevel,
|
||||
ThreadedAlertSource,
|
||||
)
|
||||
from middlewared.alert.schedule import IntervalSchedule
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
PATCH_DIR = "@PATCH_DIR@"
|
||||
DISABLED_MARKER = os.path.join(PATCH_DIR, "update_alerts_disabled")
|
||||
|
||||
_TAG_RE = re.compile(r"^v\d+\.\d+\.\d+$")
|
||||
_VERSION_RE = re.compile(r'^VERSION="([^"]+)"', re.M)
|
||||
|
||||
#: owner/repo out of any of:
|
||||
#: git@github.com:sudolulo/repo.git
|
||||
#: https://github.com/sudolulo/repo.git
|
||||
#: ssh://git@git.onetick.ninja:55214/flan/repo.git
|
||||
#: https://git.onetick.ninja/flan/repo.git
|
||||
#: The SSH port is deliberately not captured: it is not the web port.
|
||||
_REMOTE_RE = re.compile(
|
||||
r"^(?:\w+://)?(?:[^@/]+@)?([^:/]+)(?::\d+)?[:/]([^/]+)/([^/]+?)(?:\.git)?/?$"
|
||||
)
|
||||
|
||||
_TIMEOUT = 20
|
||||
|
||||
|
||||
class TrueCloudPatchUpdateAlertClass(AlertClass):
|
||||
category = AlertCategory.SYSTEM
|
||||
level = AlertLevel.INFO
|
||||
title = "truecloud-patch update available"
|
||||
text = (
|
||||
"truecloud-patch %(current)s is installed; %(latest)s is available.%(summary)s "
|
||||
"Update with: bash %(dir)s/update.sh"
|
||||
)
|
||||
|
||||
|
||||
class TrueCloudPatchSecurityUpdateAlertClass(AlertClass):
|
||||
category = AlertCategory.SYSTEM
|
||||
level = AlertLevel.WARNING
|
||||
title = "truecloud-patch security update available"
|
||||
text = (
|
||||
"truecloud-patch %(current)s is installed; %(latest)s contains a SECURITY "
|
||||
"fix.%(summary)s Update with: bash %(dir)s/update.sh"
|
||||
)
|
||||
|
||||
|
||||
class TrueCloudPatchUpdateAlertSource(ThreadedAlertSource):
|
||||
schedule = IntervalSchedule(datetime.timedelta(hours=24))
|
||||
run_on_backup_node = False
|
||||
|
||||
def check_sync(self):
|
||||
try:
|
||||
return self._check()
|
||||
except Exception:
|
||||
# An alert source must never take middlewared down with it.
|
||||
logger.debug("truecloud-patch update check failed", exc_info=True)
|
||||
return None
|
||||
|
||||
# ── internals ────────────────────────────────────────────────────────────
|
||||
|
||||
def _git(self, *args):
|
||||
# List form, never shell=True, and every `args` value is a literal from
|
||||
# this file -- nothing user-supplied reaches the command line. The partial
|
||||
# `git` path is moot: this runs as root inside middlewared, so anyone who
|
||||
# can poison PATH already has root.
|
||||
return subprocess.run( # noqa: S603
|
||||
["git", "-C", PATCH_DIR, *args], # noqa: S607
|
||||
capture_output=True, text=True, timeout=_TIMEOUT, check=True,
|
||||
).stdout
|
||||
|
||||
def _check(self):
|
||||
if os.path.exists(DISABLED_MARKER):
|
||||
return None
|
||||
if not os.path.isdir(os.path.join(PATCH_DIR, ".git")):
|
||||
return None
|
||||
|
||||
current = self._installed_version()
|
||||
if not current:
|
||||
return None
|
||||
|
||||
latest = self._latest_release_tag()
|
||||
if not latest:
|
||||
return None
|
||||
|
||||
rn = self._release_notes()
|
||||
if rn is None:
|
||||
return None
|
||||
significance, version_tuple = rn.significance, rn.version_tuple
|
||||
|
||||
if version_tuple(latest) <= version_tuple(current):
|
||||
return None
|
||||
|
||||
level, versions, summary = self._classify(
|
||||
current, latest, significance
|
||||
)
|
||||
|
||||
# Documentation-only releases are not worth an alert. This is the whole
|
||||
# point: nobody should get a notification because a README was reworded.
|
||||
if level == "docs":
|
||||
logger.debug(
|
||||
"truecloud-patch %s -> %s is documentation-only; not alerting",
|
||||
current, latest,
|
||||
)
|
||||
return None
|
||||
|
||||
args = {
|
||||
"current": f"v{current}",
|
||||
"latest": latest,
|
||||
"summary": summary,
|
||||
"dir": PATCH_DIR,
|
||||
}
|
||||
klass = (
|
||||
TrueCloudPatchSecurityUpdateAlertClass if level == "security"
|
||||
else TrueCloudPatchUpdateAlertClass
|
||||
)
|
||||
return Alert(klass, args, key=[current, latest])
|
||||
|
||||
def _release_notes(self):
|
||||
"""Load tools/release_notes.py by path.
|
||||
|
||||
NOT via sys.path: prepending would shadow the stdlib for this interpreter,
|
||||
and this runs in middlewared's thread pool, so mutating sys.path is a race.
|
||||
"""
|
||||
path = os.path.join(PATCH_DIR, "tools", "release_notes.py")
|
||||
try:
|
||||
spec = importlib.util.spec_from_file_location("_tc_release_notes", path)
|
||||
mod = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(mod)
|
||||
return mod
|
||||
except Exception:
|
||||
logger.debug("could not load release_notes", exc_info=True)
|
||||
return None
|
||||
|
||||
def _installed_version(self):
|
||||
"""The version of the patch actually checked out here."""
|
||||
try:
|
||||
with open(os.path.join(PATCH_DIR, "patch", "apply.sh"), encoding="utf-8") as fh:
|
||||
m = _VERSION_RE.search(fh.read())
|
||||
except OSError:
|
||||
return None
|
||||
return m.group(1) if m else None
|
||||
|
||||
def _latest_release_tag(self):
|
||||
"""Newest plain vX.Y.Z tag on the remote. Read-only: no .git writes.
|
||||
|
||||
Pre-release tags (-rc, -beta) are excluded: git's version sort ranks
|
||||
v0.5.0-rc1 above v0.5.0, so including them would advertise a release
|
||||
candidate as the latest stable.
|
||||
"""
|
||||
try:
|
||||
out = self._git("ls-remote", "--tags", "--refs", "origin")
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
tags = []
|
||||
for line in out.splitlines():
|
||||
parts = line.split("refs/tags/")
|
||||
if len(parts) == 2 and _TAG_RE.match(parts[1].strip()):
|
||||
tags.append(parts[1].strip())
|
||||
if not tags:
|
||||
return None
|
||||
|
||||
return max(tags, key=lambda t: tuple(int(x) for x in t.lstrip("v").split(".")))
|
||||
|
||||
def _classify(self, current, latest, significance):
|
||||
"""(level, versions, one-line summary). Falls back to alerting."""
|
||||
text = self._remote_changelog(latest)
|
||||
if text is None:
|
||||
# Cannot tell whether it matters. Alert rather than risk hiding a
|
||||
# security fix -- but say that we could not tell.
|
||||
return "notable", [], " (could not read the changelog)"
|
||||
|
||||
level, versions, headings = significance(text, current, latest)
|
||||
if level == "docs":
|
||||
return level, versions, ""
|
||||
|
||||
seen, ordered = set(), []
|
||||
for h in headings:
|
||||
if h not in seen:
|
||||
seen.add(h)
|
||||
ordered.append(h.capitalize())
|
||||
detail = ", ".join(ordered)
|
||||
return level, versions, f" Changes: {detail}." if detail else ""
|
||||
|
||||
def _changelog_url(self, tag):
|
||||
"""Where to read CHANGELOG.md at `tag`, derived from the origin remote.
|
||||
|
||||
Forge-agnostic on purpose. This project is canonically hosted on Gitea and
|
||||
mirrored to GitHub, and hard-coding either one has a nastier failure than it
|
||||
looks: when the changelog cannot be read, _classify() falls back to
|
||||
"notable" and alerts ANYWAY, because the alternative is silently hiding a
|
||||
security fix. So a stale URL does not disable the alert -- it makes the
|
||||
alert fire on every release including documentation-only ones, which is
|
||||
precisely the nagging this whole mechanism exists to prevent.
|
||||
"""
|
||||
try:
|
||||
remote = self._git("remote", "get-url", "origin").strip()
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
m = _REMOTE_RE.match(remote)
|
||||
if not m:
|
||||
return None
|
||||
|
||||
host, owner, repo = m.group(1), m.group(2), m.group(3)
|
||||
|
||||
if host.endswith("github.com"):
|
||||
return f"https://raw.githubusercontent.com/{owner}/{repo}/{tag}/CHANGELOG.md"
|
||||
|
||||
# Gitea and Forgejo both serve /{owner}/{repo}/raw/tag/{tag}/{path} over the
|
||||
# web port, which is not the SSH port the remote may name.
|
||||
return f"https://{host}/{owner}/{repo}/raw/tag/{tag}/CHANGELOG.md"
|
||||
|
||||
def _remote_changelog(self, tag):
|
||||
"""CHANGELOG.md at `tag`, over HTTPS. None if it cannot be read."""
|
||||
url = self._changelog_url(tag)
|
||||
if not url:
|
||||
return None
|
||||
try:
|
||||
with urllib.request.urlopen(url, timeout=_TIMEOUT) as resp: # noqa: S310
|
||||
if resp.status != 200:
|
||||
return None
|
||||
return resp.read().decode("utf-8", "replace")
|
||||
except Exception:
|
||||
return None
|
||||
+357
-46
@@ -32,7 +32,7 @@
|
||||
# Derive PATCH_DIR from this script's location (parent of the patch/ directory).
|
||||
PATCH_DIR="$(cd "$(dirname "$0")/.." && pwd)"
|
||||
LOG="$PATCH_DIR/apply.log"
|
||||
VERSION="0.4.0"
|
||||
VERSION="0.6.1"
|
||||
|
||||
# Rotate log at 512 KB to avoid unbounded growth on a system volume.
|
||||
# Keep two prior generations (.1 and .2) so the last three boots are always available.
|
||||
@@ -204,7 +204,91 @@ else
|
||||
_NESTED_ENABLED=0
|
||||
fi
|
||||
|
||||
# Is either module still doing something useful?
|
||||
# ── compatibility preflight ──────────────────────────────────────────────────
|
||||
#
|
||||
# The native probes above ask "has iX made this module unnecessary?". This asks the
|
||||
# other question, the dangerous one: "has iX changed middleware so that this module
|
||||
# no longer WORKS?"
|
||||
#
|
||||
# middlewared is internal API with no stability contract, and TrueNAS 26 rewrites
|
||||
# the entire cloud_backup path from async to synchronous. Every block the nested
|
||||
# module injects is an `async def` wrapping an `await`ed original; on 26 that hands
|
||||
# sync.py a coroutine where it unpacks a tuple. The backup does not fail cleanly --
|
||||
# it fails at the point where you needed it.
|
||||
#
|
||||
# tools/compat.py records what each module assumes and checks it against the
|
||||
# middlewared ACTUALLY INSTALLED HERE. A module whose assumptions no longer hold is
|
||||
# not applied. Stock TrueNAS without a feature beats TrueNAS with a broken one.
|
||||
#
|
||||
# Fail direction, deliberately asymmetric:
|
||||
# * a definite "assumption violated" -> disable that module. Strong evidence.
|
||||
# * the checker cannot run at all -> change nothing. That is a tooling glitch,
|
||||
# not evidence, and turning it into a disabled module would break working boxes.
|
||||
_TC_COMPAT_JSON="$PATCH_DIR/incompatible.json"
|
||||
rm -f "$_TC_COMPAT_JSON"
|
||||
|
||||
_tc_incompatible=0
|
||||
_tc_compat=unknown
|
||||
|
||||
# No middlewared directory means the checker has nothing to read -- every module
|
||||
# would look "broken" because every file is missing, which is the strongest possible
|
||||
# evidence derived from the weakest possible input. Skip the preflight entirely and
|
||||
# let the existing "Cannot determine middlewared directory" path handle it.
|
||||
if [ -z "$_MW_DIR" ]; then
|
||||
echo "NOTICE: middlewared directory unknown; skipping the compatibility preflight."
|
||||
_tc_compat=$(printf 'unknown\nunknown\n')
|
||||
else
|
||||
_tc_compat=$("$PYTHON" - "$PATCH_DIR" "$_MW_DIR" "$_TC_COMPAT_JSON" 2>/dev/null <<'PYEOF' || printf 'unknown\nunknown\n'
|
||||
import json, os, sys
|
||||
|
||||
patch_dir, mw_dir, out_path = sys.argv[1], sys.argv[2], sys.argv[3]
|
||||
|
||||
# APPEND, never insert(0) -- see the note by the mw_patch import below. Shadowing
|
||||
# the stdlib for this interpreter is a much worse failure than not finding compat.
|
||||
sys.path.append(os.path.join(patch_dir, 'tools'))
|
||||
try:
|
||||
import compat
|
||||
result = compat.check_tree(mw_dir)
|
||||
except Exception:
|
||||
print('unknown')
|
||||
print('unknown')
|
||||
raise SystemExit(0)
|
||||
|
||||
def verdict(r):
|
||||
# Deliberately NOT exempting 'native' here, unlike the CI matrix.
|
||||
#
|
||||
# 'native' answers "do we still NEED this module?"; 'ok' answers "is it still
|
||||
# SAFE to inject?". They are different questions, and letting native mask a
|
||||
# broken assumption conflates them: a future TrueNAS that both reworded the
|
||||
# nesting guard (-> native) AND changed the signatures (-> broken) would read
|
||||
# as safe, and we would patch it anyway.
|
||||
#
|
||||
# Refusing to apply is the correct action for BOTH answers -- a native module
|
||||
# is unnecessary and a broken one is dangerous -- so the apply path only has to
|
||||
# ask whether the assumptions hold. Whether the feature went native is decided
|
||||
# separately, by the probes above, and only affects the wording of the notice.
|
||||
return 'ok' if r['ok'] else 'broken'
|
||||
|
||||
broken = {m: r for m, r in result.items() if verdict(r) == 'broken'}
|
||||
if broken:
|
||||
# The alert source reads this. Written before we print, so a box that is
|
||||
# incompatible always has the evidence on disk even if apply.sh dies later.
|
||||
try:
|
||||
with open(out_path, 'w', encoding='utf-8') as fh:
|
||||
json.dump(broken, fh, indent=2)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
print(verdict(result['providers']))
|
||||
print(verdict(result['nested']))
|
||||
PYEOF
|
||||
)
|
||||
fi
|
||||
|
||||
_tc_compat_providers=$(printf '%s' "$_tc_compat" | sed -n '1p')
|
||||
_tc_compat_nested=$(printf '%s' "$_tc_compat" | sed -n '2p')
|
||||
|
||||
# Is either module still doing something useful -- and can it still be applied?
|
||||
_providers_needed=1
|
||||
[ "$_tc_native_b2" = "yes" ] && _providers_needed=0
|
||||
|
||||
@@ -213,6 +297,65 @@ if [ "$_NESTED_ENABLED" = "1" ] && [ "$_tc_native_nested" != "yes" ]; then
|
||||
_nested_needed=1
|
||||
fi
|
||||
|
||||
# The native checks above have already zeroed _*_needed for anything TrueNAS now
|
||||
# does itself, and printed the (good) news. Only complain about a module that is
|
||||
# still NEEDED and no longer fits -- otherwise a version that took a feature native
|
||||
# AND reshaped the module would be announced as "NOT COMPATIBLE", which is alarming
|
||||
# and false.
|
||||
if [ "$_tc_compat_providers" = "broken" ] && [ "$_providers_needed" = "1" ]; then
|
||||
echo "WARNING: truecloud-patch is NOT COMPATIBLE with this TrueNAS version."
|
||||
echo "WARNING: The B2/S3 providers module will NOT be applied. TrueCloud is"
|
||||
echo "WARNING: left stock, so B2/S3 tasks will not run until this is fixed."
|
||||
echo "WARNING: Details: $_TC_COMPAT_JSON"
|
||||
_providers_needed=0
|
||||
_tc_incompatible=1
|
||||
fi
|
||||
|
||||
if [ "$_tc_compat_nested" = "broken" ] && [ "$_nested_needed" = "1" ]; then
|
||||
echo "WARNING: truecloud-patch's nested-snapshot module is NOT COMPATIBLE with"
|
||||
echo "WARNING: this TrueNAS version and will NOT be applied. Backups still"
|
||||
echo "WARNING: run; datasets nested under the target are not included."
|
||||
echo "WARNING: Details: $_TC_COMPAT_JSON"
|
||||
_nested_needed=0
|
||||
_tc_incompatible=1
|
||||
fi
|
||||
|
||||
_tc_unmount_overlays() {
|
||||
for _tag in mw ui; do
|
||||
if mount | grep -qF "truecloud-${_tag} on "; then
|
||||
_mnt=$(mount | grep "truecloud-${_tag} on " | awk '{print $3}' | head -1)
|
||||
if umount "$_mnt" 2>/dev/null; then
|
||||
echo "NOTICE: Unmounted overlay on $_mnt"
|
||||
fi
|
||||
fi
|
||||
done
|
||||
}
|
||||
|
||||
# INCOMPATIBLE is not the same as RETIRED, and must never take the same exit.
|
||||
#
|
||||
# The kill switch below is permanent -- apply.sh checks for it and returns early on
|
||||
# every future boot -- and only install.sh removes it, NOT update.sh. That is right
|
||||
# for retirement ("TrueNAS does this natively now; stop forever"), and catastrophic
|
||||
# for incompatibility: on TrueNAS 26 the providers module fails its assumptions and
|
||||
# nested is opt-out by default, so BOTH would be zero, the kill switch would fire,
|
||||
# and the very release that fixes 26 could never re-enable itself. The user would
|
||||
# run `bash update.sh` -- exactly what the update alert tells them to do -- and the
|
||||
# patch would stay dead, silently, with their B2 backups off.
|
||||
#
|
||||
# So: incompatible means "apply nothing THIS boot, and try again next boot". The
|
||||
# fix ships, update.sh checks it out, the next boot re-runs the preflight, the
|
||||
# assumptions hold, and the patch comes back by itself.
|
||||
if [ "$_tc_incompatible" = "1" ] && [ "$_providers_needed" = "0" ] && [ "$_nested_needed" = "0" ]; then
|
||||
echo "NOTICE: Nothing can be applied on this TrueNAS version — see the WARNINGs above."
|
||||
echo "NOTICE: The kill switch is deliberately NOT set: this is an incompatibility,"
|
||||
echo "NOTICE: not a retirement. Install a release that supports this TrueNAS"
|
||||
echo "NOTICE: bash $PATCH_DIR/update.sh"
|
||||
echo "NOTICE: and the patch will re-apply itself on the next boot."
|
||||
_tc_unmount_overlays
|
||||
echo "=== done ==="
|
||||
exit 0
|
||||
fi
|
||||
|
||||
if [ "$_providers_needed" = "0" ] && [ "$_nested_needed" = "0" ]; then
|
||||
echo "NOTICE: Nothing left for truecloud-patch to do:"
|
||||
[ "$_tc_native_b2" = "yes" ] && echo "NOTICE: - TrueNAS now provides native B2 restic support."
|
||||
@@ -225,12 +368,7 @@ if [ "$_providers_needed" = "0" ] && [ "$_nested_needed" = "0" ]; then
|
||||
echo "NOTICE: Run the following to fully remove the patch:"
|
||||
echo "NOTICE: bash $PATCH_DIR/uninstall.sh"
|
||||
touch "$PATCH_DIR/disabled"
|
||||
for _tag in mw ui; do
|
||||
if mount | grep -qF "truecloud-${_tag} on "; then
|
||||
_mnt=$(mount | grep "truecloud-${_tag} on " | awk '{print $3}' | head -1)
|
||||
umount "$_mnt" 2>/dev/null && echo "NOTICE: Unmounted overlay on $_mnt" || true
|
||||
fi
|
||||
done
|
||||
_tc_unmount_overlays
|
||||
echo "=== done ==="
|
||||
exit 0
|
||||
fi
|
||||
@@ -353,19 +491,62 @@ else:
|
||||
# the guard removed but the traversal missing -- that would be a silently empty
|
||||
# backup, the worst possible outcome.
|
||||
|
||||
SNAPSHOT_BLOCK = """
|
||||
# ── nested blocks: one core, two wrappers ─────────────────────────────────────
|
||||
#
|
||||
# TrueNAS <= 25.10 has an ASYNC cloud_backup path; TrueNAS 26 rewrote it SYNCHRONOUS
|
||||
# (`middleware.call_sync` throughout, no awaits). An `async def` wrapper on 26 hands
|
||||
# sync.py a coroutine where it unpacks a tuple, and a `def` wrapper on 25.10 blocks
|
||||
# the event loop. So each block is assembled from:
|
||||
#
|
||||
# * a CORE, written once, synchronous, using middleware.call_sync -- which is safe
|
||||
# from a worker thread and deadlocks on the event loop; and
|
||||
# * a WRAPPER matching the stock function's own flavour, chosen at apply time by
|
||||
# reading whether the installed middlewared declares it `async def`.
|
||||
#
|
||||
# On <= 25.10 the async wrapper hops to a thread via `await middleware.run_in_thread`
|
||||
# -- exactly the thread call_sync needs. On 26 the stock function is already running
|
||||
# in middlewared's thread pool (its own code calls call_sync), so the sync wrapper
|
||||
# calls the core directly.
|
||||
#
|
||||
# The logic that matters -- snapshots, bind mounts, failure modes -- exists once.
|
||||
# An async twin would mean every future fix had to land twice, and the one that got
|
||||
# missed would be the one that eats a backup.
|
||||
|
||||
_NESTED_IMPORT = """
|
||||
# TRUECLOUD_PATCH — added by truenas-truecloud-patch/patch/apply.sh
|
||||
try:
|
||||
from middlewared.plugins.cloud import _truecloud_nested as _tc_nested
|
||||
except ImportError:
|
||||
_tc_nested = None
|
||||
"""
|
||||
|
||||
SNAPSHOT_CORE = _NESTED_IMPORT + """
|
||||
if _tc_nested is not None:
|
||||
_tc_orig_create_snapshot = create_snapshot
|
||||
|
||||
async def create_snapshot(middleware, path, name="cloud_task-onetime"):
|
||||
# Stock takes the (already recursive) snapshot; we only replace the PATH.
|
||||
snapshot, snap_path = await _tc_orig_create_snapshot(middleware, path, name)
|
||||
def _tc_stage(middleware, path, name, snapshot, snap_path):
|
||||
# Synchronous, and always called from a worker thread (see above).
|
||||
#
|
||||
# ONLY cloud_backup. Bail out before touching anything otherwise.
|
||||
#
|
||||
# create_snapshot is module-global in plugins/cloud/snapshot.py and is
|
||||
# imported by cloud_sync.py as well as cloud_backup/sync.py -- so this
|
||||
# wrapper sits in the path of every rclone/Storj CloudSync task with
|
||||
# snapshot=true, not just ours. Two consequences, and the second is worse:
|
||||
#
|
||||
# * everything below is a NEW failure mode for tasks that worked before we
|
||||
# were installed. A `zfs.dataset.query` that errors would break a
|
||||
# CloudSync job we have no business touching.
|
||||
# * if a CloudSync task ever were staged, nothing would ever tear it down:
|
||||
# the teardown is wired into cloud_backup's restic_backup finally, and
|
||||
# CRUD_BLOCK deliberately leaves CloudSync's nesting guard intact. The
|
||||
# bind mounts would pin the snapshot forever.
|
||||
#
|
||||
# cloud_backup names its snapshot "cloud_backup-<id>"; cloud_sync names it
|
||||
# "cloud_sync-<id>"; the stock default is "cloud_task-onetime". Anything that
|
||||
# is not ours gets stock behaviour, untouched, with no extra middleware call.
|
||||
if not name.startswith("cloud_backup"):
|
||||
return snapshot, snap_path
|
||||
|
||||
_logger = getattr(middleware, "logger", None)
|
||||
try:
|
||||
@@ -375,10 +556,13 @@ if _tc_nested is not None:
|
||||
# our staging plan would not -- silently omitting it from the backup.
|
||||
# Read afterwards, an unsnapshotted dataset instead trips the isdir()
|
||||
# check in plan_staging and fails the run loudly. Loud beats silent.
|
||||
datasets = await middleware.call(
|
||||
datasets = middleware.call_sync(
|
||||
"zfs.dataset.query", [["type", "=", "FILESYSTEM"]]
|
||||
)
|
||||
dataset, nested = get_dataset_recursive(datasets, path)
|
||||
# OUR copy of get_dataset_recursive, not the host module's: TrueNAS 26
|
||||
# deleted that helper (create_snapshot uses filesystem.statfs now), so
|
||||
# calling it out of the module namespace is a NameError there.
|
||||
dataset, nested = _tc_nested.get_dataset_recursive(datasets, path)
|
||||
|
||||
if not nested:
|
||||
# No children: stock behaviour, untouched. Stock's `finally` owns
|
||||
@@ -386,38 +570,42 @@ if _tc_nested is not None:
|
||||
# because a non-nested snapshot has no children).
|
||||
return snapshot, snap_path
|
||||
|
||||
staging_root = await _tc_nested.stage_nested(
|
||||
staging_root = _tc_nested.stage_nested(
|
||||
middleware, path, snapshot,
|
||||
dataset["name"], dataset["properties"]["mountpoint"]["value"],
|
||||
name, datasets, logger=_logger,
|
||||
)
|
||||
except Exception:
|
||||
# The snapshot exists, but this exception means sync.py never completes
|
||||
# `snapshot, local_path = await create_snapshot(...)`, so its local
|
||||
# `snapshot` stays None and its `finally` deletes NOTHING. Sweep the
|
||||
# tree ourselves or leak the parent plus one snapshot per descendant
|
||||
# dataset (160+ here) on every failed run.
|
||||
await _tc_nested.delete_snapshot_tree(middleware, snapshot, logger=_logger)
|
||||
# `snapshot, local_path = create_snapshot(...)`, so its local `snapshot`
|
||||
# stays None and its `finally` deletes NOTHING. Sweep the tree ourselves
|
||||
# or leak the parent plus one snapshot per descendant dataset (160+ here)
|
||||
# on every failed run.
|
||||
_tc_nested.delete_snapshot_tree(middleware, snapshot, logger=_logger)
|
||||
raise
|
||||
|
||||
return snapshot, staging_root
|
||||
"""
|
||||
|
||||
SNAPSHOT_ASYNC = SNAPSHOT_CORE + """
|
||||
async def create_snapshot(middleware, path, name="cloud_task-onetime"):
|
||||
snapshot, snap_path = await _tc_orig_create_snapshot(middleware, path, name)
|
||||
return await middleware.run_in_thread(
|
||||
_tc_stage, middleware, path, name, snapshot, snap_path
|
||||
)
|
||||
|
||||
create_snapshot._truecloud_patched = True
|
||||
"""
|
||||
|
||||
CRUD_BLOCK = """
|
||||
# TRUECLOUD_PATCH — added by truenas-truecloud-patch/patch/apply.sh
|
||||
try:
|
||||
from middlewared.plugins.cloud import _truecloud_nested as _tc_nested
|
||||
except ImportError:
|
||||
_tc_nested = None
|
||||
SNAPSHOT_SYNC = SNAPSHOT_CORE + """
|
||||
def create_snapshot(middleware, path, name="cloud_task-onetime"):
|
||||
snapshot, snap_path = _tc_orig_create_snapshot(middleware, path, name)
|
||||
return _tc_stage(middleware, path, name, snapshot, snap_path)
|
||||
|
||||
if _tc_nested is not None:
|
||||
_tc_orig_validate = CloudTaskServiceMixin._validate
|
||||
|
||||
async def _tc_validate(self, app, verrors, name, data):
|
||||
await _tc_orig_validate(self, app, verrors, name, data)
|
||||
create_snapshot._truecloud_patched = True
|
||||
"""
|
||||
|
||||
_CRUD_FILTER = """
|
||||
# Only cloud_backup: staging teardown is wired into cloud_backup.sync's
|
||||
# finally. cloudsync would leak bind mounts, so leave its guard intact.
|
||||
if getattr(getattr(self, "_config", None), "namespace", "") != "cloud_backup":
|
||||
@@ -438,25 +626,69 @@ if _tc_nested is not None:
|
||||
CloudTaskServiceMixin._validate._truecloud_patched = True
|
||||
"""
|
||||
|
||||
SYNC_BLOCK = """
|
||||
# TRUECLOUD_PATCH — added by truenas-truecloud-patch/patch/apply.sh
|
||||
try:
|
||||
from middlewared.plugins.cloud import _truecloud_nested as _tc_nested
|
||||
except ImportError:
|
||||
_tc_nested = None
|
||||
CRUD_ASYNC = _NESTED_IMPORT + """
|
||||
if _tc_nested is not None:
|
||||
_tc_orig_validate = CloudTaskServiceMixin._validate
|
||||
|
||||
async def _tc_validate(self, app, verrors, name, data):
|
||||
await _tc_orig_validate(self, app, verrors, name, data)
|
||||
""" + _CRUD_FILTER
|
||||
|
||||
CRUD_SYNC = _NESTED_IMPORT + """
|
||||
if _tc_nested is not None:
|
||||
_tc_orig_validate = CloudTaskServiceMixin._validate
|
||||
|
||||
def _tc_validate(self, app, verrors, name, data):
|
||||
_tc_orig_validate(self, app, verrors, name, data)
|
||||
""" + _CRUD_FILTER
|
||||
|
||||
# *args/**kwargs, not the stock signature spelled out.
|
||||
#
|
||||
# 24.10 and 25.04 have `restic_backup(middleware, job, cloud_backup, dry_run)`;
|
||||
# 25.10 added `rate_limit`. Naming them and forwarding all five raised
|
||||
# `TypeError: takes 4 positional arguments but 5 were given` on every nested backup
|
||||
# on the two older releases. Forwarding whatever we were handed makes this wrapper
|
||||
# indifferent to iX adding or dropping a trailing parameter -- which they have now
|
||||
# done twice.
|
||||
#
|
||||
# Our bind mounts pin the ZFS snapshot, so stock's `finally` cannot destroy it
|
||||
# (EBUSY) and logs one benign warning. We unmount here and then delete it for real.
|
||||
SYNC_ASYNC = _NESTED_IMPORT + """
|
||||
if _tc_nested is not None:
|
||||
_tc_orig_restic_backup = restic_backup
|
||||
|
||||
async def restic_backup(middleware, job, cloud_backup, dry_run=False, rate_limit=None):
|
||||
# Our bind mounts pin the ZFS snapshot, so stock's `finally` cannot
|
||||
# destroy it (EBUSY) and logs one benign warning. We unmount here and
|
||||
# then delete the snapshot for real.
|
||||
async def restic_backup(middleware, job, cloud_backup, *args, **kwargs):
|
||||
try:
|
||||
return await _tc_orig_restic_backup(middleware, job, cloud_backup, dry_run, rate_limit)
|
||||
return await _tc_orig_restic_backup(middleware, job, cloud_backup, *args, **kwargs)
|
||||
finally:
|
||||
try:
|
||||
await _tc_nested.cleanup_task(
|
||||
# logger= must be passed here too, exactly as the sync variant does.
|
||||
# run_in_thread forwards **kwargs (functools.partial), and without it
|
||||
# cleanup_task gets logger=None -- so every teardown warning ("could
|
||||
# not unmount X") is silently swallowed on <= 25.10, which is most
|
||||
# boxes. The two wrappers must differ ONLY in how they reach the core.
|
||||
await middleware.run_in_thread(
|
||||
_tc_nested.cleanup_task,
|
||||
middleware,
|
||||
f"cloud_backup-{cloud_backup.get('id', 'onetime')}",
|
||||
logger=getattr(middleware, "logger", None),
|
||||
)
|
||||
except Exception as e:
|
||||
middleware.logger.warning("truecloud-patch: staging cleanup failed: %r", e)
|
||||
|
||||
restic_backup._truecloud_patched = True
|
||||
"""
|
||||
|
||||
SYNC_SYNC = _NESTED_IMPORT + """
|
||||
if _tc_nested is not None:
|
||||
_tc_orig_restic_backup = restic_backup
|
||||
|
||||
def restic_backup(middleware, job, cloud_backup, *args, **kwargs):
|
||||
try:
|
||||
return _tc_orig_restic_backup(middleware, job, cloud_backup, *args, **kwargs)
|
||||
finally:
|
||||
try:
|
||||
_tc_nested.cleanup_task(
|
||||
middleware,
|
||||
f"cloud_backup-{cloud_backup.get('id', 'onetime')}",
|
||||
logger=getattr(middleware, "logger", None),
|
||||
@@ -559,10 +791,39 @@ else:
|
||||
if missing:
|
||||
raise FileNotFoundError('missing: ' + ', '.join(missing))
|
||||
|
||||
# Which flavour of cloud_backup is installed? <= 25.10 is async; TrueNAS 26
|
||||
# rewrote it synchronous. Inject the wrapper that matches: an `async def` on
|
||||
# 26 hands sync.py a coroutine where it unpacks a tuple, and a plain `def` on
|
||||
# 25.10 blocks the event loop.
|
||||
#
|
||||
# None means the three stock functions disagree, or one could not be read.
|
||||
# Refuse rather than guess -- a half-converted middleware is one this patch
|
||||
# has never seen, and guessing wrong there costs a backup, not a feature.
|
||||
# nested_src is <repo>/patch/truecloud_nested.py, so tools/ is its sibling.
|
||||
# APPEND, never insert(0) -- shadowing the stdlib for this interpreter is a
|
||||
# far worse failure than not finding compat.
|
||||
sys.path.append(
|
||||
os.path.join(os.path.dirname(os.path.dirname(nested_src)), 'tools')
|
||||
)
|
||||
import compat
|
||||
_flavour = compat.async_flavour_tree(mw_dir)
|
||||
if _flavour is None:
|
||||
raise RuntimeError(
|
||||
'cannot tell whether this TrueNAS cloud_backup path is async or '
|
||||
'sync (the wrapped functions disagree, or could not be read)'
|
||||
)
|
||||
|
||||
_snapshot_block = SNAPSHOT_ASYNC if _flavour else SNAPSHOT_SYNC
|
||||
_sync_block = SYNC_ASYNC if _flavour else SYNC_SYNC
|
||||
_crud_block = CRUD_ASYNC if _flavour else CRUD_SYNC
|
||||
|
||||
shutil.copyfile(nested_src, nested_dst) # 1. traversal implementation
|
||||
patch_file(snapshot_py, SNAPSHOT_BLOCK) # 2. build the staging tree
|
||||
patch_file(sync_path, SYNC_BLOCK) # 3. tear it down afterwards
|
||||
patch_file(crud_py, CRUD_BLOCK) # 4. ONLY NOW allow nested tasks
|
||||
patch_file(snapshot_py, _snapshot_block) # 2. build the staging tree
|
||||
patch_file(sync_path, _sync_block) # 3. tear it down afterwards
|
||||
patch_file(crud_py, _crud_block) # 4. ONLY NOW allow nested tasks
|
||||
|
||||
print('OK: cloud_backup is %s; injected the matching wrappers.'
|
||||
% ('async (TrueNAS <= 25.10)' if _flavour else 'synchronous (TrueNAS 26+)'))
|
||||
|
||||
nested_ok = True
|
||||
nested_detail = 'nested-dataset snapshots enabled (staging tree)'
|
||||
@@ -578,6 +839,51 @@ else:
|
||||
# that is inactive (superseded, or opt-in and off) is ok -- reporting a disabled
|
||||
# opt-in feature as FAIL would make `create_task.py verify` fail on a default
|
||||
# install. `active` says whether the module is doing anything.
|
||||
# ── update-available alert ────────────────────────────────────────────────────
|
||||
# Dropped into middlewared/alert/source/, where middlewared discovers and polls it
|
||||
# natively — no cron, no timer. It only raises an alert for releases that actually
|
||||
# changed something: a docs-only release is ignored (see tools/release_notes.py).
|
||||
alert_ok = False
|
||||
alert_detail = ''
|
||||
_patch_dir = os.path.dirname(os.path.dirname(nested_src))
|
||||
alert_src = os.path.join(os.path.dirname(nested_src), 'alert_source.py')
|
||||
alert_dst = os.path.join(mw_dir, 'alert', 'source', 'truecloud_patch_update.py')
|
||||
|
||||
if os.path.exists(os.path.join(_patch_dir, 'update_alerts_disabled')):
|
||||
alert_detail = 'disabled (update_alerts_disabled)'
|
||||
try:
|
||||
os.unlink(alert_dst)
|
||||
print('OK: Removed update alert (disabled).')
|
||||
except OSError:
|
||||
pass
|
||||
elif not os.path.exists(alert_src):
|
||||
alert_detail = 'alert_source.py not found'
|
||||
print(f'WARNING: {alert_src} missing — no update alert.')
|
||||
else:
|
||||
try:
|
||||
with open(alert_src, encoding='utf-8') as fh:
|
||||
_body = fh.read()
|
||||
|
||||
# repr() so ANY path becomes a valid Python literal -- a directory
|
||||
# containing a quote or backslash would otherwise produce a syntax error.
|
||||
_body = _body.replace('"@PATCH_DIR@"', repr(_patch_dir))
|
||||
|
||||
# COMPILE BEFORE WRITING. middlewared's alert.load() imports every file in
|
||||
# alert/source/ with NO try/except, and it runs at startup -- a module that
|
||||
# raises on import takes middlewared's setup down with it. An uninstalled
|
||||
# alert is a missing convenience; a broken one is a broken box.
|
||||
compile(_body, alert_dst, 'exec')
|
||||
|
||||
with open(alert_dst, 'w', encoding='utf-8') as fh:
|
||||
fh.write(_body)
|
||||
alert_ok = True
|
||||
alert_detail = 'update alert installed'
|
||||
print(f'OK: Installed update alert → {alert_dst}')
|
||||
except Exception as e:
|
||||
alert_detail = f'not applied: {e}'
|
||||
print(f'WARNING: could not install update alert: {e}')
|
||||
print('WARNING: no update alert; everything else is unaffected.')
|
||||
|
||||
patches = {
|
||||
'providers': {
|
||||
'ok': (not providers_needed) or bool(b2_ok and restic_ok),
|
||||
@@ -589,6 +895,11 @@ patches = {
|
||||
'active': nested_needed,
|
||||
'detail': nested_detail,
|
||||
},
|
||||
'update_alert': {
|
||||
'ok': True, # never a failure: it is a convenience, not a patch
|
||||
'active': alert_ok,
|
||||
'detail': alert_detail,
|
||||
},
|
||||
}
|
||||
payload = {'patched_at': time.strftime('%Y-%m-%dT%H:%M:%SZ', time.gmtime()), 'patches': patches}
|
||||
tmp = status_path + '.tmp'
|
||||
|
||||
@@ -52,7 +52,7 @@ import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
__version__ = "0.4.0"
|
||||
__version__ = "0.6.1"
|
||||
|
||||
_PATCH_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
_STATUS_FILE = os.path.join(_PATCH_DIR, "hook_status.json")
|
||||
|
||||
+12
-2
@@ -37,6 +37,10 @@ NESTED_RELPATHS = [
|
||||
#: The importable module the nested blocks depend on.
|
||||
NESTED_MODULE = ("plugins", "cloud", "_truecloud_nested.py")
|
||||
|
||||
#: The update-available alert source. Not a "patch" (it appends nothing to a stock
|
||||
#: file), but it is a file we install into middlewared and must therefore remove.
|
||||
ALERT_MODULE = ("alert", "source", "truecloud_patch_update.py")
|
||||
|
||||
|
||||
def patch_file(path, block):
|
||||
"""Append `block`, replacing any block we appended before. Idempotent."""
|
||||
@@ -101,8 +105,14 @@ def revert_nested(mw_dir):
|
||||
|
||||
|
||||
def revert_all(mw_dir):
|
||||
"""Undo every patch this project applies."""
|
||||
return revert(mw_dir, NESTED_RELPATHS + PROVIDER_RELPATHS, NESTED_MODULE)
|
||||
"""Undo every patch this project applies, and remove every file it installs."""
|
||||
reverted = revert(mw_dir, NESTED_RELPATHS + PROVIDER_RELPATHS, NESTED_MODULE)
|
||||
try:
|
||||
os.unlink(os.path.join(mw_dir, *ALERT_MODULE))
|
||||
reverted.append(ALERT_MODULE[-1])
|
||||
except OSError:
|
||||
pass
|
||||
return reverted
|
||||
|
||||
|
||||
def find_middlewared_dir():
|
||||
|
||||
+430
-41
@@ -55,9 +55,11 @@ Therefore this module owns the whole lifecycle:
|
||||
from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
import datetime
|
||||
import os
|
||||
import stat
|
||||
import subprocess
|
||||
import time
|
||||
|
||||
__all__ = [
|
||||
"STAGING_BASE",
|
||||
@@ -67,10 +69,13 @@ __all__ = [
|
||||
"cleanup_task",
|
||||
"current_mounts_under",
|
||||
"delete_snapshot_tree",
|
||||
"gc_stale_snapshots",
|
||||
"mounted_snapshots",
|
||||
"plan_staging",
|
||||
"sidecar_for",
|
||||
"snapshot_tree_names",
|
||||
"stage_nested",
|
||||
"stale_snapshot_names",
|
||||
"staging_root_for",
|
||||
"teardown",
|
||||
"verify_staged",
|
||||
@@ -113,21 +118,39 @@ def sidecar_for(staging_root: str) -> str:
|
||||
return staging_root + ".snapshot"
|
||||
|
||||
|
||||
def _write_sidecar(staging_root: str, snapshot: str) -> None:
|
||||
"""Record the pinned snapshot on disk. Blocking; call via run_in_thread."""
|
||||
def _write_sidecar(staging_root: str, snapshots) -> None:
|
||||
"""Record every snapshot tree this task still owns. One per line.
|
||||
|
||||
A LIST, not a single name -- and that is not over-engineering, it is a bug fix.
|
||||
|
||||
The sidecar used to hold one snapshot, so a run that reclaimed an older tree,
|
||||
FAILED to finish reclaiming it, and then recorded its own snapshot would
|
||||
**overwrite the only record of the survivor** -- orphaning it permanently, which is
|
||||
exactly the outcome the sidecar exists to prevent. Observed live: a snapshot
|
||||
survived one run, the next run's reclaim also failed (ZFS's 300s automount window
|
||||
had not elapsed, because the runs were minutes apart), and the record was
|
||||
destroyed anyway.
|
||||
|
||||
Now every still-pending tree is carried forward until it is actually gone.
|
||||
"""
|
||||
if isinstance(snapshots, str):
|
||||
snapshots = [snapshots]
|
||||
with contextlib.suppress(OSError):
|
||||
os.makedirs(os.path.dirname(staging_root), exist_ok=True)
|
||||
with open(sidecar_for(staging_root), "w", encoding="utf-8") as fh:
|
||||
fh.write(snapshot)
|
||||
fh.write("\n".join(dict.fromkeys(snapshots))) # de-duped, order kept
|
||||
|
||||
|
||||
def _read_sidecar(staging_root: str) -> str | None:
|
||||
"""The snapshot a previous run recorded here, if any."""
|
||||
def _read_sidecar(staging_root: str):
|
||||
"""Every snapshot tree a previous run recorded here. [] if none.
|
||||
|
||||
Tolerates the old single-line format, which is just a one-element list.
|
||||
"""
|
||||
try:
|
||||
with open(sidecar_for(staging_root), encoding="utf-8") as fh:
|
||||
return fh.read().strip() or None
|
||||
return [ln.strip() for ln in fh if ln.strip()]
|
||||
except OSError:
|
||||
return None
|
||||
return []
|
||||
|
||||
|
||||
def _remove_sidecar(staging_root: str) -> None:
|
||||
@@ -157,6 +180,76 @@ def snapshot_tree_names(snapshot: str, all_names) -> list[str]:
|
||||
]
|
||||
|
||||
|
||||
#: A snapshot must be at least this old before the garbage collector will touch it.
|
||||
#:
|
||||
#: The GC identifies our leftovers by NAME, so its only real risk is deleting a
|
||||
#: snapshot belonging to a run that is still starting up -- the window between
|
||||
#: `zfs snapshot -r` and the bind mounts appearing, which is seconds. An hour is three
|
||||
#: orders of magnitude more slack than that window needs, and still reclaims a lost
|
||||
#: tree on the very next daily run.
|
||||
GC_MIN_AGE_SECONDS = 3600
|
||||
|
||||
|
||||
def stale_snapshot_names(task_name, current_snapshot, all_names, now,
|
||||
in_use=(), min_age=GC_MIN_AGE_SECONDS):
|
||||
"""Snapshots THIS task created in an earlier run and never cleaned up.
|
||||
|
||||
Pure, because this is the one function here that DELETES DATA on a name match, and
|
||||
a name match is a weaker claim than a recorded fact. Everything it relies on is an
|
||||
argument, so every way it could be wrong is a test.
|
||||
|
||||
Why a garbage collector exists at all, when there is already a sidecar: **the
|
||||
sidecar lives in /run, which is tmpfs.** A reboot mid-backup destroys it, and with
|
||||
it the only record of a 250-snapshot tree. The sidecar handles the normal case
|
||||
precisely; this handles the case where the record itself is gone.
|
||||
|
||||
A snapshot is ours to collect only if ALL of these hold:
|
||||
|
||||
* its name is exactly ``<dataset>@<task_name>-<YYYYMMDDHHMMSS>`` -- so
|
||||
``cloud_backup-5`` never matches ``cloud_backup-50``'s snapshots, and never
|
||||
matches a periodic ``auto-2026-…`` or anything a human made;
|
||||
* it is not the snapshot the current run is using;
|
||||
* nothing is mounted from it (`in_use`) -- an in-flight run pins its own
|
||||
snapshots, so this alone protects a concurrent one-time backup;
|
||||
* it is older than `min_age` -- which covers the seconds-long window in which a
|
||||
run has taken its snapshot but not yet mounted it.
|
||||
|
||||
`now` is a timezone-aware datetime; timestamps in the name are UTC (stock builds
|
||||
them with `utc_now()`).
|
||||
"""
|
||||
prefix = task_name + "-"
|
||||
stale = []
|
||||
|
||||
for name in all_names:
|
||||
_dataset, _, snapname = name.partition("@")
|
||||
if not snapname or not snapname.startswith(prefix):
|
||||
continue
|
||||
if name == current_snapshot or snapname == _snapname_of(current_snapshot):
|
||||
continue
|
||||
if name in in_use:
|
||||
continue
|
||||
|
||||
stamp = snapname[len(prefix):]
|
||||
try:
|
||||
when = datetime.datetime.strptime(stamp, "%Y%m%d%H%M%S").replace(
|
||||
tzinfo=datetime.UTC
|
||||
)
|
||||
except ValueError:
|
||||
# Not our timestamp format. Something else owns this name; leave it alone.
|
||||
continue
|
||||
|
||||
if (now - when).total_seconds() < min_age:
|
||||
continue
|
||||
|
||||
stale.append(name)
|
||||
|
||||
return stale
|
||||
|
||||
|
||||
def _snapname_of(snapshot):
|
||||
return snapshot.partition("@")[2] if snapshot else ""
|
||||
|
||||
|
||||
def _probe_snapdir(path):
|
||||
"""Classify a snapshot directory: ``ok``, ``missing``, or why it is unusable.
|
||||
|
||||
@@ -362,18 +455,135 @@ def teardown(staging_root, runner=_run, mounts_file="/proc/self/mounts"):
|
||||
return errors
|
||||
|
||||
|
||||
# ── async orchestration (middleware is duck-typed; no middlewared import) ─────
|
||||
def snapdir_automounts(snapshot_name, mounts_file="/proc/self/mounts"):
|
||||
"""Every ``<dataset>/.zfs/snapshot/<snap>`` ZFS automount for this snapshot."""
|
||||
suffix = "/.zfs/snapshot/" + snapshot_name
|
||||
found = []
|
||||
try:
|
||||
with open(mounts_file, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
parts = line.split()
|
||||
if len(parts) > 1:
|
||||
mp = parts[1].replace("\\040", " ")
|
||||
if mp.endswith(suffix):
|
||||
found.append(mp)
|
||||
except OSError:
|
||||
return []
|
||||
return sorted(found, key=_depth, reverse=True) # deepest first
|
||||
|
||||
|
||||
async def delete_snapshot_tree(middleware, snapshot, logger=None):
|
||||
def release_snapdirs(snapshot_name, runner=_run, mounts_file="/proc/self/mounts"):
|
||||
"""Unmount ZFS's OWN snapshot automounts, so the snapshots can be destroyed.
|
||||
|
||||
Reading anything under ``<dataset>/.zfs/snapshot/<snap>/`` makes ZFS **automount**
|
||||
that snapshot, and it stays mounted for ``zfs_expire_snapshot`` seconds (300 by
|
||||
default) after the last access. teardown() unmounts OUR bind mounts -- but the
|
||||
automount underneath them survives, and while it exists ``zfs destroy`` refuses
|
||||
with *"dataset is busy"*.
|
||||
|
||||
Proven on a real pool: a 256-snapshot recursive tree swept cleanly except for the
|
||||
three datasets restic had read most recently. Those failed with EBUSY, and because
|
||||
cleanup_task removed the sidecar anyway, they were orphaned **permanently** -- a
|
||||
small leak, but a growing one, and exactly the failure this module exists to
|
||||
prevent.
|
||||
|
||||
Deepest first, so a child's automount is released before its parent's.
|
||||
"""
|
||||
errors = []
|
||||
for mp in snapdir_automounts(snapshot_name, mounts_file=mounts_file):
|
||||
res = runner(["umount", mp])
|
||||
if res.returncode != 0:
|
||||
errors.append(f"{mp}: {(res.stderr or '').strip()}")
|
||||
return errors
|
||||
|
||||
|
||||
# ── orchestration (middleware is duck-typed; no middlewared import) ───────────
|
||||
#
|
||||
# These are SYNCHRONOUS and talk to middlewared via `middleware.call_sync`, which
|
||||
# is safe from a worker thread and deadlocks on the event loop. That is the whole
|
||||
# reason this file has one implementation instead of two:
|
||||
#
|
||||
# TrueNAS <= 25.10 cloud_backup is async. The injected wrapper is `async def` and
|
||||
# hands these to `await middleware.run_in_thread(...)`, which is
|
||||
# exactly the thread `call_sync` needs.
|
||||
# TrueNAS >= 26 cloud_backup is synchronous and already runs in middlewared's
|
||||
# thread pool (its own code calls `call_sync`). The injected
|
||||
# wrapper calls these directly.
|
||||
#
|
||||
# So the async/sync difference lives entirely in the three injected blocks, and the
|
||||
# logic below -- the part with the snapshots, the bind mounts and the failure modes
|
||||
# -- is written once. Duplicating it as an async twin would mean every future fix
|
||||
# had to be made twice, and the one that got missed would be the one that eats a
|
||||
# backup.
|
||||
|
||||
|
||||
def get_dataset_recursive(datasets, directory):
|
||||
"""The dataset containing `directory`, and whether anything is nested under it.
|
||||
|
||||
Vendored from middlewared's own plugins/cloud/snapshot.py (TrueNAS <= 25.10),
|
||||
because TrueNAS 26 DELETED it -- create_snapshot there uses filesystem.statfs
|
||||
instead. The injected block used to call it out of the host module's namespace,
|
||||
which on 26 is a straight NameError.
|
||||
|
||||
Carrying our own copy removes the dependency on both versions rather than adding
|
||||
an assumption about it. It is ~10 lines of pure list arithmetic over data we
|
||||
already have in hand, and it has no reason to change.
|
||||
|
||||
Returns (dataset, has_children):
|
||||
dataset -- the DEEPEST dataset whose mountpoint is a prefix of `directory`
|
||||
has_children -- whether any OTHER dataset is mounted beneath `directory`
|
||||
"""
|
||||
datasets = [
|
||||
dict(dataset, prefixlen=len(
|
||||
os.path.dirname(os.path.commonprefix(
|
||||
[dataset["properties"]["mountpoint"]["value"] + "/", directory + "/"]))
|
||||
))
|
||||
for dataset in datasets
|
||||
if dataset["properties"]["mountpoint"]["value"] != "none"
|
||||
]
|
||||
|
||||
dataset = sorted(
|
||||
[
|
||||
dataset
|
||||
for dataset in datasets
|
||||
if (directory + "/").startswith(dataset["properties"]["mountpoint"]["value"] + "/")
|
||||
],
|
||||
key=lambda dataset: dataset["prefixlen"],
|
||||
reverse=True,
|
||||
)[0]
|
||||
|
||||
return dataset, any(
|
||||
(ds["properties"]["mountpoint"]["value"] + "/").startswith(directory + "/")
|
||||
for ds in datasets
|
||||
if ds != dataset
|
||||
)
|
||||
|
||||
|
||||
def delete_snapshot_tree(middleware, snapshot, logger=None, attempts=4,
|
||||
sleep=time.sleep):
|
||||
"""Delete the parent snapshot AND every child created by ``zfs snapshot -r``.
|
||||
|
||||
Returns the snapshots it could NOT delete -- callers must not throw that away.
|
||||
|
||||
``zfs.snapshot.delete`` is non-recursive by default and stock calls it with
|
||||
no options, so relying on stock would orphan one snapshot per descendant
|
||||
dataset on every run. Idempotent: tolerates the parent already being gone
|
||||
(stock's ``finally`` may have won the race once our mounts were released).
|
||||
|
||||
"dataset is busy" is EXPECTED here and is TRANSIENT. ZFS automounts
|
||||
``<dataset>/.zfs/snapshot/<snap>`` when it is read and keeps it mounted for
|
||||
``zfs_expire_snapshot`` seconds (300 by default) afterwards. So the datasets restic
|
||||
touched last are still pinned when we try to destroy them. We release the
|
||||
automounts explicitly and then retry -- on a real 256-snapshot tree, exactly three
|
||||
snapshots hit this, and before the fix they were orphaned permanently.
|
||||
"""
|
||||
dataset = snapshot.partition("@")[0]
|
||||
dataset, _, snapname = snapshot.partition("@")
|
||||
|
||||
# Release ZFS's own automounts first, or `zfs destroy` refuses with EBUSY on
|
||||
# everything restic read in the last few minutes.
|
||||
for err in release_snapdirs(snapname):
|
||||
if logger:
|
||||
logger.debug("truecloud-patch: could not release snapdir %s", err)
|
||||
|
||||
# Fast path: ONE recursive delete removes the parent and every child that
|
||||
# `zfs snapshot -r` created (252 on a real pool). Deleting them individually
|
||||
@@ -381,8 +591,8 @@ async def delete_snapshot_tree(middleware, snapshot, logger=None):
|
||||
# through 252 sequential deletes leaves exactly the orphans this function
|
||||
# exists to prevent.
|
||||
try:
|
||||
await middleware.call("zfs.snapshot.delete", snapshot, {"recursive": True})
|
||||
return
|
||||
middleware.call_sync("zfs.snapshot.delete", snapshot, {"recursive": True})
|
||||
return []
|
||||
except Exception as e: # noqa: BLE001 - fall through to the explicit sweep
|
||||
# Usually just "parent already gone" (stock's finally won the race once our
|
||||
# mounts were released), which the sweep below handles. Log it rather than
|
||||
@@ -398,7 +608,7 @@ async def delete_snapshot_tree(middleware, snapshot, logger=None):
|
||||
# our mounts are released -- which fails the recursive delete while the
|
||||
# children survive. Sweep them by name.
|
||||
try:
|
||||
snaps = await middleware.call(
|
||||
snaps = middleware.call_sync(
|
||||
"zfs.snapshot.query", [["name", "^", dataset]], {"select": ["name"]}
|
||||
)
|
||||
# An empty result means the tree is already gone -- delete nothing, and
|
||||
@@ -413,17 +623,130 @@ async def delete_snapshot_tree(middleware, snapshot, logger=None):
|
||||
)
|
||||
names = [snapshot]
|
||||
|
||||
for name in names:
|
||||
def confirm_gone(failed):
|
||||
"""Drop any name ZFS no longer has, even though its delete raised.
|
||||
|
||||
A delete that raised "does not exist" SUCCEEDED as far as we care, and must
|
||||
not be retried or reported. The query is only a refinement: if it cannot be
|
||||
answered we keep the delete's own verdict, rather than inventing survivors --
|
||||
a false survivor keeps the sidecar forever and is reported as a leak that
|
||||
isn't there.
|
||||
"""
|
||||
if not failed:
|
||||
return []
|
||||
try:
|
||||
await middleware.call("zfs.snapshot.delete", name)
|
||||
except Exception as e: # noqa: BLE001 - already gone is fine
|
||||
live = middleware.call_sync(
|
||||
"zfs.snapshot.query", [["name", "^", dataset]], {"select": ["name"]}
|
||||
)
|
||||
except Exception: # noqa: BLE001 - cannot refine; trust the delete's verdict
|
||||
return list(failed)
|
||||
live = {s["name"] for s in live}
|
||||
return [n for n in failed if n in live]
|
||||
|
||||
remaining = list(names)
|
||||
for attempt in range(attempts):
|
||||
failed = []
|
||||
for name in remaining:
|
||||
try:
|
||||
middleware.call_sync("zfs.snapshot.delete", name)
|
||||
except Exception: # noqa: BLE001 - busy, or already gone; sorted out below
|
||||
failed.append(name)
|
||||
|
||||
remaining = confirm_gone(failed)
|
||||
if not remaining:
|
||||
return []
|
||||
|
||||
if attempt < attempts - 1:
|
||||
# EBUSY is the automount expiring. Release again (anything that walks
|
||||
# .zfs can re-automount a snapshot) and give it a moment.
|
||||
release_snapdirs(snapname)
|
||||
sleep(5)
|
||||
|
||||
for name in remaining:
|
||||
if logger:
|
||||
logger.warning(
|
||||
"truecloud-patch: could not delete snapshot %s: %r", name, e
|
||||
"truecloud-patch: could not delete snapshot %s after %d attempts "
|
||||
"(still busy?) -- it will be reclaimed on the next run",
|
||||
name, attempts,
|
||||
)
|
||||
return remaining
|
||||
|
||||
|
||||
def mounted_snapshots(mounts_file="/proc/self/mounts"):
|
||||
"""Every ZFS snapshot something is currently mounted from.
|
||||
|
||||
The device field of a snapshot mount IS the snapshot name (`Tap/apps/x@snap`), for
|
||||
both our staging bind mounts and ZFS's own .zfs automounts. So this is a direct,
|
||||
factual answer to "is anything using this snapshot right now" -- which is what
|
||||
protects a concurrently-running backup from the garbage collector, rather than
|
||||
trusting an age heuristic to be generous enough.
|
||||
"""
|
||||
live = set()
|
||||
try:
|
||||
with open(mounts_file, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
dev = line.split(" ", 1)[0]
|
||||
if "@" in dev:
|
||||
live.add(dev.replace("\\040", " "))
|
||||
except OSError:
|
||||
return set()
|
||||
return live
|
||||
|
||||
|
||||
def gc_stale_snapshots(middleware, task_name, current_snapshot, logger=None,
|
||||
now=None, mounts_file="/proc/self/mounts"):
|
||||
"""Delete snapshots this task left behind in an earlier run. Returns what remains.
|
||||
|
||||
The backstop for when the RECORD is gone, not just the snapshots: the sidecar lives
|
||||
in /run (tmpfs), so a reboot mid-backup takes it with them. Without this, that tree
|
||||
-- one snapshot per descendant dataset, 250+ on a real pool -- is orphaned with
|
||||
nothing left pointing at it.
|
||||
|
||||
Selection is `stale_snapshot_names()`, which is pure and heavily tested, because a
|
||||
name match is a weaker claim than a recorded fact and this deletes data on one.
|
||||
"""
|
||||
dataset = current_snapshot.partition("@")[0]
|
||||
now = now or datetime.datetime.now(datetime.UTC)
|
||||
|
||||
try:
|
||||
snaps = middleware.call_sync(
|
||||
"zfs.snapshot.query", [["name", "^", dataset]], {"select": ["name"]}
|
||||
)
|
||||
except Exception as e: # noqa: BLE001 - cannot enumerate; collect nothing
|
||||
if logger:
|
||||
logger.warning(
|
||||
"truecloud-patch: could not enumerate snapshots for GC: %r", e
|
||||
)
|
||||
return []
|
||||
|
||||
stale = stale_snapshot_names(
|
||||
task_name, current_snapshot, [s["name"] for s in snaps], now,
|
||||
in_use=mounted_snapshots(mounts_file),
|
||||
)
|
||||
if not stale:
|
||||
return []
|
||||
|
||||
if logger:
|
||||
logger.warning(
|
||||
"truecloud-patch: %d snapshot(s) from an earlier run of %s were never "
|
||||
"cleaned up (a lost record, e.g. a reboot mid-backup); collecting them",
|
||||
len(stale), task_name,
|
||||
)
|
||||
|
||||
remaining = []
|
||||
for name in stale:
|
||||
try:
|
||||
middleware.call_sync("zfs.snapshot.delete", name)
|
||||
except Exception as e: # noqa: BLE001 - busy, or gone; either way, next run
|
||||
remaining.append(name)
|
||||
if logger:
|
||||
logger.debug(
|
||||
"truecloud-patch: could not collect %s: %r", name, e
|
||||
)
|
||||
return remaining
|
||||
|
||||
async def stage_nested(middleware, path, snapshot, base_dataset, base_mountpoint,
|
||||
|
||||
def stage_nested(middleware, path, snapshot, base_dataset, base_mountpoint,
|
||||
task_name, datasets, logger=None):
|
||||
"""Build a complete staging tree for `path` from the already-taken `snapshot`.
|
||||
|
||||
@@ -447,30 +770,56 @@ async def stage_nested(middleware, path, snapshot, base_dataset, base_mountpoint
|
||||
staging_root = staging_root_for(task_name)
|
||||
|
||||
# A previous run may have crashed mid-flight; never build on top of that.
|
||||
await middleware.run_in_thread(teardown, staging_root)
|
||||
teardown(staging_root)
|
||||
|
||||
# ...and if it left a sidecar behind, that snapshot tree is still on disk and
|
||||
# nothing else will ever reclaim it. Sweep it before we overwrite the record,
|
||||
# or a single crashed run orphans 160+ snapshots permanently.
|
||||
stale = await middleware.run_in_thread(_read_sidecar, staging_root)
|
||||
if stale and stale != snapshot:
|
||||
# ...and if it left snapshot trees behind, they are still on disk and nothing else
|
||||
# will ever reclaim them. Sweep them before recording our own, or a single crashed
|
||||
# run orphans 160+ snapshots permanently.
|
||||
#
|
||||
# Anything a reclaim FAILS to delete is carried forward, not dropped. Overwriting
|
||||
# the sidecar with only our own snapshot is what destroyed the record of a survivor
|
||||
# once already: the reclaim ran, hit ZFS's 300-second automount window (the runs
|
||||
# were minutes apart), left one snapshot behind, and then the record of it was
|
||||
# overwritten -- a permanent orphan, created by the very code meant to prevent one.
|
||||
pending = []
|
||||
for stale in _read_sidecar(staging_root):
|
||||
if stale == snapshot:
|
||||
continue
|
||||
if logger:
|
||||
logger.warning(
|
||||
"truecloud-patch: reclaiming snapshot tree from an earlier "
|
||||
"interrupted run: %s", stale,
|
||||
"run: %s", stale,
|
||||
)
|
||||
pending.extend(delete_snapshot_tree(middleware, stale, logger=logger))
|
||||
|
||||
if pending and logger:
|
||||
logger.warning(
|
||||
"truecloud-patch: %d snapshot(s) from an earlier run are still busy; "
|
||||
"carrying them forward to the next run", len(pending),
|
||||
)
|
||||
|
||||
# ...and collect anything from an earlier run that has NO record at all.
|
||||
#
|
||||
# The sidecar above is precise but lives in /run, which is tmpfs -- a reboot
|
||||
# mid-backup destroys it and orphans the whole tree with nothing pointing at it.
|
||||
# This finds those by name and is the only thing that ever will.
|
||||
#
|
||||
# It runs AFTER the sidecar reclaim on purpose: the recorded path is authoritative
|
||||
# and cheap, and the GC should only ever be mopping up what the record lost.
|
||||
pending.extend(
|
||||
gc_stale_snapshots(middleware, task_name, snapshot, logger=logger)
|
||||
)
|
||||
await delete_snapshot_tree(middleware, stale, logger=logger)
|
||||
|
||||
# Record the snapshot BEFORE mounting anything, not after. middlewared can
|
||||
# die at any point (this patch even schedules a restart at boot), and the
|
||||
# sidecar is the only thing that survives it -- an in-process dict would take
|
||||
# the sole record of a 160-snapshot tree with it. Writing it after apply_plan
|
||||
# would leave exactly the crash window the sidecar exists to close.
|
||||
await middleware.run_in_thread(_write_sidecar, staging_root, snapshot)
|
||||
_write_sidecar(staging_root, [*pending, snapshot])
|
||||
|
||||
try:
|
||||
mounts, skipped = await middleware.run_in_thread(
|
||||
plan_staging, base_dataset, base_mountpoint, path, snapshot_name,
|
||||
mounts, skipped = plan_staging(
|
||||
base_dataset, base_mountpoint, path, snapshot_name,
|
||||
datasets, staging_root,
|
||||
)
|
||||
if logger:
|
||||
@@ -479,11 +828,21 @@ async def stage_nested(middleware, path, snapshot, base_dataset, base_mountpoint
|
||||
"truecloud-patch: not staging dataset %r: %s", name, reason
|
||||
)
|
||||
|
||||
await middleware.run_in_thread(apply_plan, mounts)
|
||||
await middleware.run_in_thread(verify_staged, mounts)
|
||||
apply_plan(mounts)
|
||||
verify_staged(mounts)
|
||||
except Exception:
|
||||
await middleware.run_in_thread(teardown, staging_root)
|
||||
await middleware.run_in_thread(_remove_sidecar, staging_root)
|
||||
# Tear down the mounts, but KEEP the sidecar.
|
||||
#
|
||||
# The caller (SNAPSHOT_BLOCK) sweeps the snapshot tree on the way out, and if
|
||||
# any of it is still busy it will survive -- and the sidecar is the only record
|
||||
# that it exists. Removing it here would orphan those snapshots permanently.
|
||||
#
|
||||
# The asymmetry is deliberate: a sidecar left behind when the tree is already
|
||||
# gone is harmless (the next run tries to delete a tree that is not there,
|
||||
# finds nothing, and moves on), while a sidecar removed while the tree still
|
||||
# exists is unrecoverable. Only a confirmed-clean sweep removes it -- see
|
||||
# cleanup_task().
|
||||
teardown(staging_root)
|
||||
raise
|
||||
|
||||
if logger:
|
||||
@@ -494,24 +853,55 @@ async def stage_nested(middleware, path, snapshot, base_dataset, base_mountpoint
|
||||
return staging_root
|
||||
|
||||
|
||||
async def cleanup_task(middleware, task_name, logger=None):
|
||||
def cleanup_task(middleware, task_name, logger=None):
|
||||
"""Tear down a task's staging tree and delete the snapshot it pinned.
|
||||
|
||||
Safe to call unconditionally: a no-op when the task was never staged.
|
||||
"""
|
||||
staging_root = staging_root_for(task_name)
|
||||
snapshot = _read_sidecar(staging_root)
|
||||
pinned = _read_sidecar(staging_root)
|
||||
|
||||
if snapshot is None and not os.path.isdir(staging_root):
|
||||
if not pinned and not os.path.isdir(staging_root):
|
||||
return # never staged; nothing to do
|
||||
|
||||
errors = await middleware.run_in_thread(teardown, staging_root)
|
||||
errors = teardown(staging_root)
|
||||
if errors and logger:
|
||||
for err in errors:
|
||||
logger.warning("truecloud-patch: staging teardown: %s", err)
|
||||
|
||||
if snapshot is not None:
|
||||
await delete_snapshot_tree(middleware, snapshot, logger=logger)
|
||||
if not pinned:
|
||||
_remove_sidecar(staging_root)
|
||||
return
|
||||
|
||||
# Every tree this task still owns -- ours, plus anything an earlier run could not
|
||||
# finish reclaiming.
|
||||
survivors = []
|
||||
for snapshot in pinned:
|
||||
survivors.extend(delete_snapshot_tree(middleware, snapshot, logger=logger))
|
||||
|
||||
# KEEP the sidecar if anything survived. It is the only record that those
|
||||
# snapshots exist, and removing it orphans them permanently.
|
||||
#
|
||||
# That is not theoretical: on a real 256-snapshot tree, three snapshots were still
|
||||
# pinned by ZFS's own .zfs/snapshot automount (which lingers for 300s after the
|
||||
# last read), failed to delete with "dataset is busy", and the sidecar was removed
|
||||
# anyway -- so nothing would ever have reclaimed them. A small leak, but one that
|
||||
# grows by a few snapshots on every single run, forever.
|
||||
#
|
||||
# Left in place, the next run's stage_nested() sees a stale sidecar naming a
|
||||
# different snapshot and sweeps that tree first -- by which time the automounts are
|
||||
# long gone and the delete succeeds.
|
||||
if survivors:
|
||||
if logger:
|
||||
logger.warning(
|
||||
"truecloud-patch: %d snapshot(s) could not be deleted (still busy); "
|
||||
"recording them so the next run reclaims them: %s",
|
||||
len(survivors), ", ".join(survivors),
|
||||
)
|
||||
# The SURVIVORS, not the trees we asked to delete. Writing the original list
|
||||
# back would keep re-sweeping trees that are already gone.
|
||||
_write_sidecar(staging_root, survivors)
|
||||
return
|
||||
|
||||
_remove_sidecar(staging_root)
|
||||
|
||||
@@ -540,8 +930,7 @@ def cleanup_all(base=None, runner=_run, mounts_file="/proc/self/mounts",
|
||||
# a sidecar is the only record that an interrupted run's snapshot tree (one
|
||||
# snapshot per descendant dataset) is still on disk.
|
||||
for sc in sorted(glob_fn(os.path.join(base, "*.snapshot"))):
|
||||
snap = read_sidecar(sc[: -len(".snapshot")])
|
||||
if snap:
|
||||
for snap in read_sidecar(sc[: -len(".snapshot")]):
|
||||
lines.append(f" NOTE: an interrupted backup left snapshot '{snap}' behind.")
|
||||
lines.append(f" Remove it and its children: zfs destroy -r '{snap}'")
|
||||
|
||||
|
||||
+1
-1
@@ -17,7 +17,7 @@
|
||||
# bash /mnt/tank/truenas-truecloud-patch/patch/apply.sh
|
||||
# systemctl restart middlewared
|
||||
|
||||
VERSION="0.4.0"
|
||||
VERSION="0.6.1"
|
||||
|
||||
PATCH_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
|
||||
|
||||
Executable
+332
@@ -0,0 +1,332 @@
|
||||
#!/usr/bin/env bash
|
||||
# Cut a release. Two stages, and you cannot skip the first one.
|
||||
#
|
||||
# bash release.sh 0.6.0 --rc stage 1: candidate. Invisible to users.
|
||||
# bash release.sh 0.6.0 --promote stage 2: stable. Only if an rc passed HERE.
|
||||
#
|
||||
# WHY IT WORKS THIS WAY
|
||||
#
|
||||
# This repo once cut twelve releases in a day, several of them fixing the release
|
||||
# before. Every one of those raises an update alert on every user's box. An alert
|
||||
# people learn to ignore is worse than no alert, because one day it will be
|
||||
# carrying a security fix.
|
||||
#
|
||||
# So: debugging happens across rc1, rc2, rc3 -- which update.sh and the alert
|
||||
# source both filter out, so no user ever sees them -- and a stable tag is only
|
||||
# reachable from a candidate that already went green on the identical commit.
|
||||
# tools/release_gate.py enforces that here, and .github/workflows/release.yml
|
||||
# enforces it again where it cannot be bypassed.
|
||||
#
|
||||
# Day to day you do not touch this script. You write your changes under
|
||||
# `## Unreleased` in CHANGELOG.md and push to main. Releasing is a separate,
|
||||
# deliberate act.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# This file IS on every user's box -- update.sh clones the whole repo -- so it
|
||||
# carries no VERSION= not because it is "not shipped", but because nothing reads
|
||||
# it. VERSION= exists so the running system can say which patch it is; this script
|
||||
# never runs on a running system. (Anything that DOES carry a VERSION= must be in
|
||||
# release_notes.VERSIONED_FILES or it silently rots: create_task.py sat three
|
||||
# releases behind for exactly that reason.)
|
||||
#
|
||||
# Running it on a user's box is a no-op by construction, and that is checked below
|
||||
# rather than left to luck: update.sh pins the checkout to a tag in detached HEAD,
|
||||
# and this refuses to run anywhere but an up-to-date `main` with push access.
|
||||
|
||||
cd "$(dirname "$(readlink -f "$0")")"
|
||||
|
||||
die() { printf '\033[31merror:\033[0m %s\n' "$*" >&2; exit 1; }
|
||||
note() { printf '\033[36m==>\033[0m %s\n' "$*"; }
|
||||
ok() { printf '\033[32m ok\033[0m %s\n' "$*"; }
|
||||
|
||||
usage() {
|
||||
sed -n '2,25p' "$0" | sed 's/^# \{0,1\}//'
|
||||
exit "${1:-0}"
|
||||
}
|
||||
|
||||
# ── args ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
target=""
|
||||
mode=""
|
||||
assume_yes=0
|
||||
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--rc) mode="rc" ;;
|
||||
--promote) mode="promote" ;;
|
||||
--check) mode="check" ;;
|
||||
-y|--yes) assume_yes=1 ;;
|
||||
-h|--help) usage 0 ;;
|
||||
-*) die "unknown option: $1" ;;
|
||||
*)
|
||||
[ -n "$target" ] && die "give exactly one version"
|
||||
target="${1#v}"
|
||||
;;
|
||||
esac
|
||||
shift
|
||||
done
|
||||
|
||||
[ -n "$target" ] || usage 2
|
||||
[ -n "$mode" ] || die "pick a stage: --rc (candidate) or --promote (stable)"
|
||||
|
||||
printf '%s' "$target" | grep -Eq '^[0-9]+\.[0-9]+\.[0-9]+$' \
|
||||
|| die "version must be plain X.Y.Z (the -rcN suffix is added for you)"
|
||||
|
||||
# ── preflight ────────────────────────────────────────────────────────────────
|
||||
|
||||
[ -d .git ] || die "not a git checkout"
|
||||
|
||||
# UNTRACKED files count as dirty, because --rc runs `git add -A`: an untracked file
|
||||
# lying around would be swept into the release commit and pushed. This checkout is
|
||||
# routinely shared between sessions, so stray files are the normal state here, not
|
||||
# an exotic one. (Anything genuinely ignorable belongs in .gitignore.)
|
||||
if [ -n "$(git status --porcelain)" ]; then
|
||||
echo " Working tree is not clean:" >&2
|
||||
git status --short >&2
|
||||
die "commit, stash, or ignore the above first -- a release must be reproducible
|
||||
from a commit, not from whatever happened to be on disk. --rc runs
|
||||
'git add -A', so an untracked file here ships inside the release."
|
||||
fi
|
||||
|
||||
branch="$(git rev-parse --abbrev-ref HEAD)"
|
||||
if [ "$branch" = "HEAD" ]; then
|
||||
# This is what an INSTALLED patch looks like: update.sh pins the checkout to a
|
||||
# release tag in detached HEAD. Someone has found this script in their clone and
|
||||
# run it. Say so plainly rather than emitting a confusing branch error.
|
||||
die "this is an installed checkout (detached at $(git describe --tags --always)),
|
||||
not a development one. release.sh is the maintainer tool that publishes new
|
||||
versions of the patch; it is not how you install or update one.
|
||||
|
||||
To update this box: bash update.sh"
|
||||
fi
|
||||
[ "$branch" = "main" ] || die "releases are cut from main, not '$branch'"
|
||||
|
||||
note "fetching tags"
|
||||
git fetch --tags --quiet origin
|
||||
|
||||
if [ -n "$(git log --oneline "origin/$branch..$branch" 2>/dev/null)" ]; then
|
||||
die "local main has commits that are not pushed. Push first: the tag must point
|
||||
at a commit the world can actually fetch."
|
||||
fi
|
||||
|
||||
# BEHIND is just as bad as ahead, and less obvious. --rc commits the version stamp,
|
||||
# creates the tag, and only THEN pushes -- so on a stale main the push is rejected
|
||||
# (non-fast-forward) *after* the tag exists and `## Unreleased` has already been
|
||||
# consumed. Re-running then sees rc1, cuts rc2, and silently skips the stamping
|
||||
# step; the rc1 tag dangles locally forever.
|
||||
if [ -n "$(git log --oneline "$branch..origin/$branch" 2>/dev/null)" ]; then
|
||||
die "local main is BEHIND origin. Pull first: git pull --ff-only
|
||||
Releasing from a stale main half-completes: the tag is cut locally, the push
|
||||
is rejected, and '## Unreleased' has already been consumed."
|
||||
fi
|
||||
|
||||
# ── the gates: identical to the ones CI will run ─────────────────────────────
|
||||
|
||||
run_gates() {
|
||||
local tag="$1"
|
||||
note "gate: versions agree, CHANGELOG is complete"
|
||||
python3 tools/release_notes.py check "$tag" \
|
||||
|| die "content gate failed (see above)"
|
||||
ok "content"
|
||||
|
||||
note "gate: provenance"
|
||||
python3 tools/release_gate.py "$tag" -C . \
|
||||
|| die "provenance gate failed (see above)"
|
||||
ok "provenance"
|
||||
}
|
||||
|
||||
# ── tests, because a tag that fails its own tests is not a release ───────────
|
||||
|
||||
# The interpreter that actually has the dev deps. A bare `python3` is usually the
|
||||
# system one with no pytest -- and "No module named pytest" would read as "tests
|
||||
# fail", i.e. the gate blocking a release for a reason that is not true.
|
||||
PY=python3
|
||||
[ -x .venv/bin/python ] && PY=.venv/bin/python
|
||||
|
||||
RUFF=""
|
||||
if [ -x .venv/bin/ruff ]; then RUFF=.venv/bin/ruff
|
||||
elif command -v ruff >/dev/null 2>&1; then RUFF=ruff
|
||||
fi
|
||||
|
||||
run_tests() {
|
||||
note "running the suite"
|
||||
"$PY" -c 'import pytest' 2>/dev/null || die "no pytest in $PY. Install the dev deps:
|
||||
python3 -m venv .venv && .venv/bin/pip install pytest ruff"
|
||||
"$PY" -m pytest tests -q || die "tests fail. Fix them; do not release around them."
|
||||
[ -n "$RUFF" ] && { "$RUFF" check patch tests tools || die "lint fails"; }
|
||||
local f
|
||||
while IFS= read -r f; do
|
||||
bash -n "$f" || die "bash syntax error in $f"
|
||||
done < <(find . -name '*.sh' -not -path './.git/*')
|
||||
ok "suite"
|
||||
}
|
||||
|
||||
confirm() {
|
||||
[ "$assume_yes" -eq 1 ] && return 0
|
||||
printf '\n%s [y/N] ' "$1"
|
||||
read -r reply </dev/tty
|
||||
case "$reply" in [yY]*) return 0 ;; *) die "aborted" ;; esac
|
||||
}
|
||||
|
||||
# ── check ────────────────────────────────────────────────────────────────────
|
||||
|
||||
if [ "$mode" = "check" ]; then
|
||||
echo
|
||||
python3 - "$target" <<'PY'
|
||||
import sys, os
|
||||
sys.path.insert(0, os.path.join(os.getcwd(), "tools"))
|
||||
from release_notes import unreleased_body
|
||||
with open("CHANGELOG.md", encoding="utf-8") as fh:
|
||||
body = unreleased_body(fh.read())
|
||||
if body:
|
||||
print("Unreleased, and would ship as v%s:\n" % sys.argv[1])
|
||||
print("\n".join(" " + line for line in body.splitlines()))
|
||||
else:
|
||||
print("Nothing under `## Unreleased`. There is nothing to release.")
|
||||
PY
|
||||
echo
|
||||
next_rc="$(python3 tools/release_gate.py "$target" --next-rc -C .)"
|
||||
echo "Next candidate would be: $next_rc"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# ── stage 1: release candidate ───────────────────────────────────────────────
|
||||
|
||||
if [ "$mode" = "rc" ]; then
|
||||
tag="$(python3 tools/release_gate.py "$target" --next-rc -C .)"
|
||||
|
||||
# Tests BEFORE the stamping commit, deliberately.
|
||||
#
|
||||
# Stamping consumes `## Unreleased` and makes a "release vX.Y.Z" commit. If the
|
||||
# suite then failed, that commit was already on main and a re-run died inside
|
||||
# promote() with "no `## Unreleased` content" -- the release was wedged, and the
|
||||
# only way out was to hand-unpick a commit. Failing first leaves the tree
|
||||
# untouched.
|
||||
run_tests
|
||||
|
||||
# The first candidate promotes `## Unreleased` and stamps the version into every
|
||||
# script. Later candidates (rc2+) are re-cuts of an already-stamped version, so
|
||||
# they only tag -- the CHANGELOG section for this version already exists, and
|
||||
# fixes found during rc go into it.
|
||||
#
|
||||
# "Already stamped" is decided by the TREE, not by the rc1 tag: if a previous run
|
||||
# stamped and committed but died before tagging (or before pushing), the tag is
|
||||
# absent while the stamp is present, and re-stamping would try to promote an
|
||||
# `## Unreleased` section that is no longer there.
|
||||
if python3 tools/release_notes.py check "v$target-rc0" >/dev/null 2>&1; then
|
||||
note "v$target is already stamped; cutting a follow-up candidate"
|
||||
else
|
||||
note "promoting '## Unreleased' -> v$target and stamping the scripts"
|
||||
python3 - "$target" <<'PY'
|
||||
import datetime, os, re, sys
|
||||
sys.path.insert(0, os.path.join(os.getcwd(), "tools"))
|
||||
from release_notes import VERSIONED_FILES, promote
|
||||
|
||||
version = sys.argv[1]
|
||||
today = datetime.date.today().isoformat()
|
||||
|
||||
with open("CHANGELOG.md", encoding="utf-8") as fh:
|
||||
text = fh.read()
|
||||
try:
|
||||
out = promote(text, version, today)
|
||||
except ValueError as e:
|
||||
sys.exit(f"error: {e}")
|
||||
with open("CHANGELOG.md", "w", encoding="utf-8") as fh:
|
||||
fh.write(out)
|
||||
print(f" CHANGELOG.md Unreleased -> v{version} - {today}")
|
||||
|
||||
for rel in VERSIONED_FILES:
|
||||
with open(rel, encoding="utf-8") as fh:
|
||||
src = fh.read()
|
||||
new, n = re.subn(
|
||||
r'^(VERSION=|__version__\s*=\s*)"[^"]+"',
|
||||
lambda m: f'{m.group(1)}"{version}"',
|
||||
src, count=1, flags=re.M,
|
||||
)
|
||||
if not n:
|
||||
sys.exit(f"error: {rel} has no VERSION= line to stamp")
|
||||
if new != src:
|
||||
with open(rel, "w", encoding="utf-8") as fh:
|
||||
fh.write(new)
|
||||
print(f" {rel} -> {version}")
|
||||
PY
|
||||
git add -A
|
||||
git commit -q -m "release v$target"
|
||||
ok "stamped"
|
||||
fi
|
||||
|
||||
# Gated as the rc tag it is: content is checked against the base version, and the
|
||||
# provenance gate is a no-op for candidates -- being one is the whole point.
|
||||
# (run_tests already ran, above, before anything was committed.)
|
||||
run_gates "$tag"
|
||||
|
||||
echo
|
||||
note "about to cut $tag"
|
||||
echo " commit: $(git rev-parse --short HEAD) $(git log -1 --format=%s)"
|
||||
echo
|
||||
echo " A candidate is invisible to users: update.sh and the update alert both"
|
||||
echo " ignore -rc tags. Install it on a real box, exercise it, and only then"
|
||||
echo " run: bash release.sh $target --promote"
|
||||
confirm "cut $tag?"
|
||||
|
||||
# Push the branch FIRST. If it is rejected, no tag has been created yet -- a tag
|
||||
# pointing at a commit nobody else has is worse than no tag, because the next run
|
||||
# sees it, counts it as a candidate, and cuts rc2 against a commit that was never
|
||||
# published.
|
||||
git push --quiet origin main
|
||||
git tag -a "$tag" -m "$tag"
|
||||
git push --quiet origin "$tag"
|
||||
ok "pushed $tag"
|
||||
echo
|
||||
echo "CI is now testing $tag and publishing it as a PRE-RELEASE."
|
||||
echo "When you are satisfied: bash release.sh $target --promote"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# ── stage 2: promote to stable ───────────────────────────────────────────────
|
||||
|
||||
if [ "$mode" = "promote" ]; then
|
||||
tag="v$target"
|
||||
|
||||
if git rev-parse -q --verify "refs/tags/$tag" >/dev/null; then
|
||||
next="$(echo "$target" | awk -F. '{printf "%d.%d.%d", $1, $2, $3+1}')"
|
||||
die "$tag already exists. A published version is immutable -- if it is broken,
|
||||
the fix ships as v$next, and it goes through a candidate like everything else."
|
||||
fi
|
||||
|
||||
# The barrier. Fails unless an rc points at THIS commit.
|
||||
note "gate: was this exact commit a release candidate?"
|
||||
python3 tools/release_gate.py "$tag" -C . || {
|
||||
echo
|
||||
die "not promotable (see above)"
|
||||
}
|
||||
ok "provenance"
|
||||
|
||||
run_gates "$tag"
|
||||
run_tests
|
||||
|
||||
rcs="$(python3 - "$target" <<'PY'
|
||||
import os, sys
|
||||
sys.path.insert(0, os.path.join(os.getcwd(), "tools"))
|
||||
from release_gate import rc_tags
|
||||
print(", ".join(rc_tags(sys.argv[1])) or "none")
|
||||
PY
|
||||
)"
|
||||
|
||||
echo
|
||||
note "about to publish $tag to every user"
|
||||
echo " commit: $(git rev-parse --short HEAD)"
|
||||
echo " candidates: $rcs"
|
||||
echo
|
||||
echo " This raises an update alert on every installed box (unless the only"
|
||||
echo " CHANGELOG section is Docs). Make sure it is worth interrupting people."
|
||||
confirm "publish $tag?"
|
||||
|
||||
git tag -a "$tag" -m "$tag"
|
||||
git push --quiet origin "$tag"
|
||||
ok "pushed $tag"
|
||||
echo
|
||||
echo "CI is publishing the release. Users will be alerted within 24h."
|
||||
exit 0
|
||||
fi
|
||||
@@ -0,0 +1,132 @@
|
||||
"""Tests for the update-available alert source.
|
||||
|
||||
This file is imported by middlewared's `alert.load()`, which runs at STARTUP and
|
||||
has **no try/except**:
|
||||
|
||||
def load(self):
|
||||
for module in load_modules(.../alert/source):
|
||||
for cls in load_classes(module, AlertSource, (ThreadedAlertSource,)):
|
||||
source = cls(self.middleware)
|
||||
if source.name in ALERT_SOURCES:
|
||||
raise RuntimeError(...)
|
||||
|
||||
So a module that raises on import takes middlewared's setup down with it. These
|
||||
tests guard the realistic ways that could happen.
|
||||
"""
|
||||
|
||||
import ast
|
||||
import os
|
||||
import re
|
||||
|
||||
import pytest
|
||||
|
||||
ALERT_SRC = os.path.join(os.path.dirname(__file__), "..", "patch", "alert_source.py")
|
||||
APPLY_SH = os.path.join(os.path.dirname(__file__), "..", "patch", "apply.sh")
|
||||
|
||||
|
||||
def source():
|
||||
with open(ALERT_SRC, encoding="utf-8") as fh:
|
||||
return fh.read()
|
||||
|
||||
|
||||
def tree():
|
||||
return ast.parse(source())
|
||||
|
||||
|
||||
class TestCannotBreakMiddlewaredAtImport:
|
||||
def test_it_compiles(self):
|
||||
compile(source(), "alert_source.py", "exec")
|
||||
|
||||
def test_apply_sh_compiles_it_before_writing_it(self):
|
||||
# The substituted file is what middlewared imports. If it does not compile,
|
||||
# installing it would break startup — so apply.sh must refuse to write it.
|
||||
with open(APPLY_SH, encoding="utf-8") as fh:
|
||||
sh = fh.read()
|
||||
i = sh.index("alert_dst = os.path.join(mw_dir, 'alert', 'source'")
|
||||
block = sh[i:i + 1600]
|
||||
assert "compile(_body, alert_dst, 'exec')" in block
|
||||
assert block.index("compile(_body") < block.index("open(alert_dst, 'w'")
|
||||
|
||||
def test_patch_dir_is_substituted_with_repr(self):
|
||||
# A directory containing a quote or backslash would otherwise produce a
|
||||
# syntax error in the installed module.
|
||||
with open(APPLY_SH, encoding="utf-8") as fh:
|
||||
sh = fh.read()
|
||||
assert "_body.replace('\"@PATCH_DIR@\"', repr(_patch_dir))" in sh
|
||||
|
||||
@pytest.mark.parametrize("path", [
|
||||
"/mnt/tank/patch",
|
||||
'/mnt/we"ird/patch', # a quote in the path
|
||||
"/mnt/back\\slash/patch", # a backslash
|
||||
])
|
||||
def test_substituted_module_compiles_for_awkward_paths(self, path):
|
||||
body = source().replace('"@PATCH_DIR@"', repr(path))
|
||||
compile(body, "alert_source.py", "exec") # must not raise
|
||||
|
||||
def test_no_io_at_module_import_time(self):
|
||||
# Anything at module scope runs during alert.load(). Only imports,
|
||||
# constants and class definitions are allowed.
|
||||
allowed = (ast.Import, ast.ImportFrom, ast.Assign, ast.AnnAssign,
|
||||
ast.ClassDef, ast.FunctionDef, ast.Expr)
|
||||
for node in tree().body:
|
||||
assert isinstance(node, allowed), f"module-level {type(node).__name__}"
|
||||
if isinstance(node, ast.Expr):
|
||||
assert isinstance(node.value, ast.Constant), "only the docstring"
|
||||
|
||||
|
||||
class TestAlertClassNaming:
|
||||
"""middlewared's AlertClassMeta raises NameError unless the name ends in
|
||||
'AlertClass' — at import, inside alert.load(), which has no try/except."""
|
||||
|
||||
def alert_classes(self):
|
||||
return [n for n in tree().body
|
||||
if isinstance(n, ast.ClassDef)
|
||||
and any(getattr(b, "id", "") == "AlertClass" for b in n.bases)]
|
||||
|
||||
def test_there_are_alert_classes(self):
|
||||
assert self.alert_classes()
|
||||
|
||||
def test_every_alert_class_name_ends_in_AlertClass(self):
|
||||
for cls in self.alert_classes():
|
||||
assert cls.name.endswith("AlertClass"), (
|
||||
f"{cls.name}: AlertClassMeta raises NameError on this"
|
||||
)
|
||||
|
||||
def test_every_alert_class_defines_the_required_attrs(self):
|
||||
# category/level/title are NotImplemented on the base; a missing one shows
|
||||
# up as a broken alert rather than an error.
|
||||
for cls in self.alert_classes():
|
||||
names = {t.id for n in cls.body if isinstance(n, ast.Assign)
|
||||
for t in n.targets if isinstance(t, ast.Name)}
|
||||
assert {"category", "level", "title", "text"} <= names, cls.name
|
||||
|
||||
def test_alert_text_placeholders_match_the_args_we_pass(self):
|
||||
src = source()
|
||||
placeholders = set(re.findall(r"%\((\w+)\)s", src))
|
||||
# These are the keys built in _check().
|
||||
assert placeholders <= {"current", "latest", "summary", "dir"}
|
||||
|
||||
|
||||
class TestNoSysPathMutation:
|
||||
def test_release_notes_is_loaded_by_path_not_sys_path(self):
|
||||
# sys.path.insert(0, ...) would shadow the stdlib for this interpreter, and
|
||||
# ThreadedAlertSource runs in a thread pool — mutating sys.path is a race.
|
||||
#
|
||||
# Check for actual MUTATION, not the string: the docstring legitimately
|
||||
# mentions sys.path to explain why it is avoided.
|
||||
src = source()
|
||||
assert "sys.path.insert" not in src
|
||||
assert "sys.path.append" not in src
|
||||
assert not re.search(r"^import sys$", src, re.M), "sys is not needed"
|
||||
assert "spec_from_file_location" in src
|
||||
|
||||
|
||||
class TestNeverWritesToGit:
|
||||
def test_only_read_only_git_commands(self):
|
||||
# A `git fetch` from middlewared (running as root) would leave root-owned
|
||||
# objects in .git and break every later non-root git command — which is
|
||||
# exactly the breakage this project already hit once.
|
||||
src = source()
|
||||
for forbidden in ("fetch", "pull", "checkout", "clone", "reset"):
|
||||
assert f'"{forbidden}"' not in src, f"git {forbidden} writes to .git"
|
||||
assert '"ls-remote"' in src
|
||||
+159
-23
@@ -14,14 +14,27 @@ import pytest
|
||||
|
||||
APPLY_SH = os.path.join(os.path.dirname(__file__), "..", "patch", "apply.sh")
|
||||
|
||||
#: Every block that is actually injected into a middlewared module.
|
||||
#:
|
||||
#: The three nested blocks come in two flavours. TrueNAS <= 25.10 has an ASYNC
|
||||
#: cloud_backup path; TrueNAS 26 rewrote it synchronous. apply.sh reads which one is
|
||||
#: installed and injects the matching wrapper -- an `async def` on 26 would hand
|
||||
#: sync.py a coroutine where it unpacks a tuple, and a plain `def` on 25.10 would
|
||||
#: block the event loop. Both flavours must therefore be valid Python, always.
|
||||
EXPECTED_BLOCKS = {
|
||||
"B2_BLOCK",
|
||||
"RESTIC_BLOCK",
|
||||
"SNAPSHOT_BLOCK",
|
||||
"CRUD_BLOCK",
|
||||
"SYNC_BLOCK",
|
||||
"SNAPSHOT_ASYNC",
|
||||
"SNAPSHOT_SYNC",
|
||||
"CRUD_ASYNC",
|
||||
"CRUD_SYNC",
|
||||
"SYNC_ASYNC",
|
||||
"SYNC_SYNC",
|
||||
}
|
||||
|
||||
NESTED_BLOCKS = ["SNAPSHOT_ASYNC", "SNAPSHOT_SYNC", "CRUD_ASYNC", "CRUD_SYNC",
|
||||
"SYNC_ASYNC", "SYNC_SYNC"]
|
||||
|
||||
|
||||
def heredoc_source():
|
||||
with open(APPLY_SH, encoding="utf-8") as fh:
|
||||
@@ -32,18 +45,30 @@ def heredoc_source():
|
||||
|
||||
|
||||
def extract_blocks():
|
||||
"""The blocks as apply.sh actually builds them.
|
||||
|
||||
EVALUATED, not read off as string literals: each nested block is a CORE
|
||||
concatenated with a flavour-specific wrapper, so reading only `ast.Constant`
|
||||
would silently return nothing for them -- a green suite over blocks nobody
|
||||
checked. Assignments that need the runtime (argv, imports) simply fail to
|
||||
evaluate and are skipped.
|
||||
"""
|
||||
tree = ast.parse(heredoc_source())
|
||||
blocks = {}
|
||||
for node in ast.walk(tree):
|
||||
if isinstance(node, ast.Assign):
|
||||
ns, blocks = {}, {}
|
||||
for node in tree.body:
|
||||
if not isinstance(node, ast.Assign):
|
||||
continue
|
||||
try:
|
||||
value = eval( # noqa: S307 - our own shipped source, on purpose
|
||||
compile(ast.Expression(node.value), "<blocks>", "eval"), {}, ns
|
||||
)
|
||||
except Exception:
|
||||
continue
|
||||
for tgt in node.targets:
|
||||
if (
|
||||
isinstance(tgt, ast.Name)
|
||||
and tgt.id.endswith("_BLOCK")
|
||||
and isinstance(node.value, ast.Constant)
|
||||
and isinstance(node.value.value, str)
|
||||
):
|
||||
blocks[tgt.id] = node.value.value
|
||||
if isinstance(tgt, ast.Name) and isinstance(value, str):
|
||||
ns[tgt.id] = value
|
||||
if tgt.id in EXPECTED_BLOCKS:
|
||||
blocks[tgt.id] = value
|
||||
return blocks
|
||||
|
||||
|
||||
@@ -98,7 +123,7 @@ def test_injected_block_carries_the_idempotency_marker(name):
|
||||
assert extract_blocks()[name].lstrip("\n").startswith("# TRUECLOUD_PATCH")
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name", ["SNAPSHOT_BLOCK", "CRUD_BLOCK", "SYNC_BLOCK"])
|
||||
@pytest.mark.parametrize("name", NESTED_BLOCKS)
|
||||
def test_nested_blocks_degrade_safely_without_the_module(name):
|
||||
# If _truecloud_nested failed to install, every nested block must no-op.
|
||||
# Critically this includes CRUD_BLOCK: relaxing the guard without the
|
||||
@@ -119,20 +144,21 @@ class TestSnapshotLeak:
|
||||
# On a staging failure, sync.py's `snapshot, local_path = await
|
||||
# create_snapshot(...)` never completes, so its local `snapshot` stays
|
||||
# None and its finally deletes nothing. We must sweep it ourselves.
|
||||
block = extract_blocks()["SNAPSHOT_BLOCK"]
|
||||
block = extract_blocks()["SNAPSHOT_ASYNC"]
|
||||
assert "except Exception:" in block
|
||||
assert "delete_snapshot_tree" in block
|
||||
assert "raise" in block
|
||||
|
||||
def test_sync_block_cleans_up_on_every_path(self):
|
||||
block = extract_blocks()["SYNC_BLOCK"]
|
||||
block = extract_blocks()["SYNC_ASYNC"]
|
||||
assert "finally:" in block
|
||||
assert "cleanup_task" in block
|
||||
|
||||
|
||||
def test_crud_block_is_scoped_to_cloud_backup():
|
||||
# cloudsync has no staging teardown wired in, so its guard must stay.
|
||||
assert '!= "cloud_backup"' in extract_blocks()["CRUD_BLOCK"]
|
||||
for name in ("CRUD_ASYNC", "CRUD_SYNC"):
|
||||
assert '!= "cloud_backup"' in extract_blocks()[name]
|
||||
|
||||
|
||||
class TestIndependentModules:
|
||||
@@ -220,7 +246,7 @@ class TestIndependentModules:
|
||||
# find the string in our own patch and never detect native support.
|
||||
sh = self._sh()
|
||||
assert "split('\\n# TRUECLOUD_PATCH', 1)[0]" in sh
|
||||
assert "no further nesting" in extract_blocks()["CRUD_BLOCK"], (
|
||||
assert "no further nesting" in extract_blocks()["CRUD_ASYNC"], (
|
||||
"if this ever stops being true, the probe comment is stale"
|
||||
)
|
||||
|
||||
@@ -264,7 +290,7 @@ class TestOptIn:
|
||||
# The guard-relaxing crud.py patch must be inside the enabled branch.
|
||||
src = heredoc_source()
|
||||
gate = src.index("if not nested_needed:")
|
||||
crud = src.index("patch_file(crud_py, CRUD_BLOCK)")
|
||||
crud = src.index("patch_file(crud_py, _crud_block)")
|
||||
assert gate < crud, "crud.py patch must sit inside the opt-in branch"
|
||||
|
||||
def test_disabling_REVERTS_the_patch_rather_than_merely_skipping_it(self):
|
||||
@@ -282,7 +308,7 @@ class TestOptIn:
|
||||
assert "from mw_patch import patch_file, revert_nested" in src
|
||||
gate = src.index("if not nested_needed:")
|
||||
revert = src.index("reverted = revert_nested(")
|
||||
patch = src.index("patch_file(crud_py, CRUD_BLOCK)")
|
||||
patch = src.index("patch_file(crud_py, _crud_block)")
|
||||
assert gate < revert < patch, "revert belongs in the not-needed branch"
|
||||
|
||||
def test_import_failure_skips_the_patch_rather_than_crashing(self):
|
||||
@@ -302,8 +328,118 @@ def test_guard_is_relaxed_only_after_traversal_is_installed():
|
||||
src = heredoc_source()
|
||||
order = [
|
||||
src.index("shutil.copyfile(nested_src, nested_dst)"),
|
||||
src.index("patch_file(snapshot_py, SNAPSHOT_BLOCK)"),
|
||||
src.index("patch_file(sync_path, SYNC_BLOCK)"),
|
||||
src.index("patch_file(crud_py, CRUD_BLOCK)"),
|
||||
src.index("patch_file(snapshot_py, _snapshot_block)"),
|
||||
src.index("patch_file(sync_path, _sync_block)"),
|
||||
src.index("patch_file(crud_py, _crud_block)"),
|
||||
]
|
||||
assert order == sorted(order), "crud.py must be patched last"
|
||||
|
||||
|
||||
class TestWrappersDoNotHardcodeStockArity:
|
||||
"""iX changes the tail of these signatures between releases.
|
||||
|
||||
SYNC_BLOCK used to spell out `(middleware, job, cloud_backup, dry_run, rate_limit)`
|
||||
and forward all five. But 24.10 and 25.04 declare only four -- `rate_limit` arrived
|
||||
in 25.10 -- so every nested backup on those two releases raised
|
||||
`TypeError: restic_backup() takes 4 positional arguments but 5 were given`.
|
||||
It shipped broken and nothing noticed, because the compat check at the time only
|
||||
asked whether the parameter NAMES still appeared somewhere in the signature.
|
||||
|
||||
Forwarding *args/**kwargs makes the wrapper indifferent to a trailing parameter
|
||||
being added or dropped, which is the only part iX actually churns.
|
||||
"""
|
||||
|
||||
def test_restic_backup_forwards_rather_than_naming_stock_params(self):
|
||||
block = extract_blocks()["SYNC_ASYNC"]
|
||||
assert "async def restic_backup(middleware, job, cloud_backup, *args, **kwargs)" in block
|
||||
assert "_tc_orig_restic_backup(middleware, job, cloud_backup, *args, **kwargs)" in block
|
||||
|
||||
# Comments stripped: the block's own commentary explains the rate_limit
|
||||
# history, and that must not be mistaken for the code re-declaring it.
|
||||
code = "\n".join(
|
||||
line for line in block.splitlines()
|
||||
if not line.lstrip().startswith("#")
|
||||
)
|
||||
assert "rate_limit" not in code, (
|
||||
"naming a trailing stock parameter re-introduces the arity bug"
|
||||
)
|
||||
|
||||
|
||||
class TestTheTwoNativeProbesCannotDrift:
|
||||
"""The split-literal trick is implemented TWICE: inline in apply.sh's probe, and
|
||||
as compat._squash. It has already caused one silent bug.
|
||||
|
||||
Stock middleware writes the guard as an implicitly-concatenated literal, so the
|
||||
contiguous phrase never appears in the source. A naive search finds nothing,
|
||||
concludes iX removed the guard, and reports "native" -- which means "retire the
|
||||
module". That would disable nested snapshots on every box that depends on them.
|
||||
|
||||
apply.sh (runtime, on the box) and compat.py (static, in CI) must therefore agree
|
||||
on every input, or one of them is wrong about whether to retire a module.
|
||||
"""
|
||||
|
||||
CASES = [
|
||||
# (crud.py source, expected native?)
|
||||
("verrors.add('x', 'datasets that have no further nesting')", False),
|
||||
# THE case: split across adjacent literals, as stock actually writes it.
|
||||
("verrors.add('x', 'datasets that have no further '\n"
|
||||
" 'nesting')", False),
|
||||
('verrors.add("x", "no further "\n "nesting")', False),
|
||||
# Guard genuinely gone -> iX implemented it -> native.
|
||||
("verrors.add('x', 'some other validation entirely')", True),
|
||||
("", True),
|
||||
]
|
||||
|
||||
@pytest.mark.parametrize("src,expect_native", CASES)
|
||||
def test_both_probes_agree(self, src, expect_native):
|
||||
import sys as _sys
|
||||
_sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "tools"))
|
||||
import compat
|
||||
|
||||
shipped = _nested_native_detector()(src)
|
||||
assert (shipped == "yes") == expect_native, (
|
||||
f"apply.sh's probe says native={shipped!r} for {src!r}"
|
||||
)
|
||||
|
||||
path, phrase, native_when_present = compat.NATIVE_PROBES[compat.NESTED]
|
||||
present = compat._squash(phrase) in compat._squash(src)
|
||||
static_native = (present == native_when_present)
|
||||
assert static_native == expect_native, (
|
||||
f"compat.py says native={static_native} for {src!r}"
|
||||
)
|
||||
|
||||
|
||||
class TestOnlyOurOwnTasksAreTouched:
|
||||
"""create_snapshot is module-global, and cloud_sync.py imports it too.
|
||||
|
||||
plugins/cloud/snapshot.py::create_snapshot is imported by BOTH
|
||||
cloud_backup/sync.py and cloud_sync.py, so our wrapper sits in the path of every
|
||||
rclone/Storj CloudSync task with snapshot=true -- tasks this patch has no business
|
||||
touching. Two consequences, the second much worse than the first:
|
||||
|
||||
* every middleware call we add is a NEW failure mode for a job that worked
|
||||
before we were installed;
|
||||
* a staged CloudSync task would NEVER be torn down. The teardown is wired into
|
||||
cloud_backup's restic_backup finally, and CRUD_BLOCK deliberately leaves
|
||||
CloudSync's nesting guard intact -- so the bind mounts would pin the ZFS
|
||||
snapshot forever.
|
||||
|
||||
cloud_backup names its snapshot "cloud_backup-<id>", cloud_sync "cloud_sync-<id>",
|
||||
and stock's default is "cloud_task-onetime".
|
||||
"""
|
||||
|
||||
@pytest.mark.parametrize("name", ["SNAPSHOT_ASYNC", "SNAPSHOT_SYNC"])
|
||||
def test_the_staging_path_is_gated_on_cloud_backup(self, name):
|
||||
block = extract_blocks()[name]
|
||||
assert 'if not name.startswith("cloud_backup"):' in block
|
||||
|
||||
@pytest.mark.parametrize("name", ["SNAPSHOT_ASYNC", "SNAPSHOT_SYNC"])
|
||||
def test_the_bail_out_precedes_every_middleware_call(self, name):
|
||||
# The point is to add NO new failure mode to a CloudSync task. If any
|
||||
# middleware call happened before the bail-out, we would already have broken
|
||||
# the thing we are trying not to touch.
|
||||
block = extract_blocks()[name]
|
||||
gate = block.index('if not name.startswith("cloud_backup"):')
|
||||
for call in ("middleware.call_sync(", "_tc_nested.stage_nested(",
|
||||
"_tc_nested.delete_snapshot_tree("):
|
||||
assert gate < block.index(call), f"{call} runs before the cloud_backup gate"
|
||||
|
||||
@@ -0,0 +1,361 @@
|
||||
"""Tests for the middlewared compatibility manifest.
|
||||
|
||||
Two failure directions, and they are NOT symmetric:
|
||||
|
||||
* a false **BROKEN** makes a module decline to apply on a box where it works.
|
||||
Worse, if both modules go quiet, apply.sh used to set a PERMANENT kill switch
|
||||
that only install.sh clears -- so a network blip or an innocent refactor could
|
||||
take a working box's B2 backups down until someone noticed by hand.
|
||||
|
||||
* a false **OK** lets the patch inject into middleware it does not fit, which is a
|
||||
broken backup discovered at restore time.
|
||||
|
||||
Both are tested. The `native` verdict gets its own scrutiny because it is the most
|
||||
dangerous thing this file can say -- it means "TrueNAS does this now, retire the
|
||||
module" -- and it rests on nothing more than a substring match.
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "tools"))
|
||||
|
||||
import compat # noqa: E402
|
||||
from compat import ( # noqa: E402
|
||||
NESTED,
|
||||
PROVIDERS,
|
||||
Unreadable,
|
||||
check,
|
||||
is_broken,
|
||||
)
|
||||
|
||||
# A middlewared that the patch fits: TrueNAS 25.10 in miniature.
|
||||
GOOD = {
|
||||
"rclone/remote/b2.py": "class B2RcloneRemote(BaseRcloneRemote):\n pass\n",
|
||||
"plugins/cloud_backup/restic.py": (
|
||||
"class ResticConfig:\n cmd: list\n\n"
|
||||
"def get_restic_config(cloud_backup):\n return ResticConfig([], {})\n"
|
||||
),
|
||||
"plugins/cloud/snapshot.py": (
|
||||
'async def create_snapshot(middleware, path, name="x"):\n return "s", "p"\n'
|
||||
),
|
||||
"plugins/cloud/crud.py": (
|
||||
"class CloudTaskServiceMixin:\n"
|
||||
" async def _validate(self, app, verrors, name, data):\n"
|
||||
" verrors.add('x', 'datasets that have no further '\n"
|
||||
" 'nesting')\n"
|
||||
),
|
||||
"plugins/cloud_backup/sync.py": (
|
||||
"async def restic_backup(middleware, job, cloud_backup, dry_run=False, "
|
||||
"rate_limit=None):\n pass\n"
|
||||
),
|
||||
# The middlewared METHODS the injected code calls. TrueNAS 26 deleted both of
|
||||
# these files, taking zfs.dataset.query / zfs.snapshot.query / zfs.snapshot.delete
|
||||
# with them -- see TestMiddlewareMethodsWeCall.
|
||||
"plugins/zfs_/dataset.py": (
|
||||
"class ZFSDataset(CRUDService):\n"
|
||||
" class Config:\n"
|
||||
" namespace = 'zfs.dataset'\n"
|
||||
" def query(self, filters, options):\n pass\n"
|
||||
),
|
||||
"plugins/zfs_/snapshot.py": (
|
||||
"class ZFSSnapshot(CRUDService):\n"
|
||||
" class Config:\n"
|
||||
" namespace = 'zfs.snapshot'\n"
|
||||
" def query(self, filters, options):\n pass\n"
|
||||
" def delete(self, id_, options={}):\n pass\n"
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def loader(files):
|
||||
def load(path):
|
||||
if path not in files:
|
||||
return None
|
||||
v = files[path]
|
||||
if isinstance(v, Exception):
|
||||
raise v
|
||||
return v
|
||||
return load
|
||||
|
||||
|
||||
def check_files(files, modules=None):
|
||||
return check(loader(files), modules)
|
||||
|
||||
|
||||
def with_(**overrides):
|
||||
files = dict(GOOD)
|
||||
files.update(overrides)
|
||||
return files
|
||||
|
||||
|
||||
class TestTheBaseline:
|
||||
def test_a_good_tree_is_ok_and_not_native(self):
|
||||
r = check_files(GOOD)
|
||||
for mod in (PROVIDERS, NESTED):
|
||||
assert r[mod]["ok"], r[mod]["problems"]
|
||||
assert not r[mod]["native"]
|
||||
assert not r[mod]["unknown"]
|
||||
|
||||
|
||||
class TestFalseOkWouldBreakBackups:
|
||||
"""The patch calls the originals POSITIONALLY. A name-subset check passed all of
|
||||
these, and each is a TypeError or -- worse -- silently swapped arguments."""
|
||||
|
||||
def test_reordered_parameters_are_broken(self):
|
||||
r = check_files(with_(**{
|
||||
"plugins/cloud/snapshot.py":
|
||||
'async def create_snapshot(name, path, middleware):\n return 1, 2\n',
|
||||
}))
|
||||
assert is_broken(r[NESTED])
|
||||
|
||||
def test_a_keyword_only_conversion_is_broken(self):
|
||||
r = check_files(with_(**{
|
||||
"plugins/cloud/snapshot.py":
|
||||
'async def create_snapshot(middleware, *, path, name="x"):\n return 1, 2\n',
|
||||
}))
|
||||
assert is_broken(r[NESTED])
|
||||
|
||||
def test_a_new_required_parameter_is_broken(self):
|
||||
r = check_files(with_(**{
|
||||
"plugins/cloud/snapshot.py":
|
||||
'async def create_snapshot(middleware, path, name, dataset):\n return 1, 2\n',
|
||||
}))
|
||||
assert is_broken(r[NESTED])
|
||||
|
||||
def test_a_new_optional_parameter_is_fine(self):
|
||||
# The patch simply will not pass it. Refusing here would be false BROKEN.
|
||||
r = check_files(with_(**{
|
||||
"plugins/cloud/snapshot.py":
|
||||
'async def create_snapshot(middleware, path, name="x", quiet=False):\n'
|
||||
" return 1, 2\n",
|
||||
}))
|
||||
assert r[NESTED]["ok"], r[NESTED]["problems"]
|
||||
|
||||
def test_the_master_signature_change_is_caught(self):
|
||||
# iX really did rename this on master: get_restic_config(entry, credentials).
|
||||
# RESTIC_BLOCK rebinds the module-level name to a 1-arg wrapper, so getting
|
||||
# this wrong kills EVERY TrueCloud task -- Storj included.
|
||||
r = check_files(with_(**{
|
||||
"plugins/cloud_backup/restic.py":
|
||||
"class ResticConfig:\n cmd: list\n\n"
|
||||
"def get_restic_config(entry, credentials):\n pass\n",
|
||||
}))
|
||||
assert is_broken(r[PROVIDERS])
|
||||
|
||||
def test_a_vanished_symbol_is_broken(self):
|
||||
r = check_files(with_(**{
|
||||
"plugins/cloud/snapshot.py": "def something_else():\n pass\n",
|
||||
}))
|
||||
assert is_broken(r[NESTED])
|
||||
|
||||
|
||||
class TestFalseBrokenWouldDisableWorkingBoxes:
|
||||
def test_a_conditionally_defined_symbol_is_not_broken(self):
|
||||
r = check_files(with_(**{
|
||||
"plugins/cloud/snapshot.py":
|
||||
"try:\n"
|
||||
" from .fast import create_snapshot\n"
|
||||
"except ImportError:\n"
|
||||
' async def create_snapshot(middleware, path, name="x"):\n'
|
||||
" return 1, 2\n",
|
||||
}))
|
||||
assert not is_broken(r[NESTED]), r[NESTED]["problems"]
|
||||
|
||||
def test_a_re_exported_symbol_is_unknown_not_broken(self):
|
||||
r = check_files(with_(**{
|
||||
"plugins/cloud_backup/restic.py":
|
||||
"from ._impl import ResticConfig, get_restic_config\n",
|
||||
}))
|
||||
assert not is_broken(r[PROVIDERS])
|
||||
assert r[PROVIDERS]["unknown"]
|
||||
|
||||
def test_an_unreadable_source_is_unknown_not_broken(self):
|
||||
# A rate limit (the matrix makes ~30 unauthenticated requests) must not be
|
||||
# able to say "iX deleted six files, both modules are broken".
|
||||
r = check_files(with_(**{
|
||||
"plugins/cloud/snapshot.py": Unreadable("HTTP 429"),
|
||||
}))
|
||||
assert not is_broken(r[NESTED])
|
||||
assert r[NESTED]["unknown"]
|
||||
|
||||
def test_a_definite_break_still_wins_over_an_unknown(self):
|
||||
r = check_files(with_(**{
|
||||
"plugins/cloud/snapshot.py": Unreadable("HTTP 429"),
|
||||
"plugins/cloud_backup/sync.py":
|
||||
"async def restic_backup(job, middleware, cloud_backup):\n pass\n",
|
||||
}))
|
||||
assert is_broken(r[NESTED]), "unknown must not launder away a proven break"
|
||||
|
||||
|
||||
class TestTheNativeVerdict:
|
||||
""""native" means "retire the module". It is the most destructive thing this file
|
||||
can say, and it is only a substring match — so it must never outrank BROKEN."""
|
||||
|
||||
def test_broken_outranks_native(self):
|
||||
# Guard reworded (reads as native) AND the signatures changed (really broken).
|
||||
# This used to render as good news: green CI, no bug report, and a README row
|
||||
# telling users the feature went native while it was in fact broken.
|
||||
r = check_files(with_(**{
|
||||
"plugins/cloud/crud.py":
|
||||
"class CloudTaskServiceMixin:\n"
|
||||
" async def _validate(self, verrors, name):\n"
|
||||
" verrors.add('x', 'no children allowed')\n",
|
||||
}))
|
||||
assert r[NESTED]["native"]
|
||||
assert is_broken(r[NESTED])
|
||||
assert compat._verdict(r[NESTED]) == "BROKEN"
|
||||
|
||||
def test_an_already_patched_tree_does_not_read_as_native(self):
|
||||
# B2_BLOCK writes `B2RcloneRemote.restic = True` into b2.py. Scanning the whole
|
||||
# file finds OUR OWN line and concludes TrueNAS went native — so the command
|
||||
# compat.py's docstring recommends for a live box (`--tree /usr/lib/...`)
|
||||
# reported providers as native on every patched machine.
|
||||
r = check_files(with_(**{
|
||||
"rclone/remote/b2.py":
|
||||
"class B2RcloneRemote(BaseRcloneRemote):\n pass\n"
|
||||
"\n# TRUECLOUD_PATCH — added by truenas-truecloud-patch/patch/apply.sh\n"
|
||||
"B2RcloneRemote.restic = True\n",
|
||||
}))
|
||||
assert not r[PROVIDERS]["native"], "read its own patch as native support"
|
||||
assert r[PROVIDERS]["ok"]
|
||||
|
||||
def test_a_genuinely_native_b2_is_native(self):
|
||||
r = check_files(with_(**{
|
||||
"rclone/remote/b2.py":
|
||||
"class B2RcloneRemote(BaseRcloneRemote):\n restic = True\n",
|
||||
}))
|
||||
assert r[PROVIDERS]["native"]
|
||||
|
||||
|
||||
class TestUpdateReadmeCannotPublishAGuess:
|
||||
def test_it_refuses_when_anything_is_unknown(self, tmp_path):
|
||||
readme = tmp_path / "README.md"
|
||||
readme.write_text(f"x\n{compat.BEGIN}\nold\n{compat.END}\ny\n")
|
||||
rows = [{
|
||||
"ref": "TS-25.10.4", "unreleased": False,
|
||||
"modules": check_files(with_(**{
|
||||
"plugins/cloud/snapshot.py": Unreadable("HTTP 429"),
|
||||
})),
|
||||
}]
|
||||
with pytest.raises(Unreadable):
|
||||
compat.update_readme(rows, path=str(readme))
|
||||
assert "old" in readme.read_text(), "a blip must not repaint the matrix"
|
||||
|
||||
|
||||
class TestAsyncFlavour:
|
||||
"""TrueNAS <= 25.10 is async; 26 is synchronous. Both are supported -- apply.sh
|
||||
injects the wrapper that matches. So asyncness is DETECTED, never assumed."""
|
||||
|
||||
def test_async_middleware_is_detected(self):
|
||||
assert compat.async_flavour(loader(GOOD)) is True
|
||||
|
||||
def test_sync_middleware_is_detected(self):
|
||||
sync = dict(GOOD)
|
||||
sync["plugins/cloud/snapshot.py"] = (
|
||||
'def create_snapshot(middleware, path, name="x"):\n return "s", "p"\n'
|
||||
)
|
||||
sync["plugins/cloud/crud.py"] = (
|
||||
"class CloudTaskServiceMixin:\n"
|
||||
" def _validate(self, app, verrors, name, data):\n"
|
||||
" verrors.add('x', 'no further nesting')\n"
|
||||
)
|
||||
sync["plugins/cloud_backup/sync.py"] = (
|
||||
"def restic_backup(middleware, job, cloud_backup, dry_run=False, "
|
||||
"rate_limit=None):\n pass\n"
|
||||
)
|
||||
assert compat.async_flavour(loader(sync)) is False
|
||||
|
||||
def test_a_HALF_converted_middleware_is_refused(self):
|
||||
# The dangerous middle. If iX converts create_snapshot but not restic_backup,
|
||||
# there is no single wrapper flavour that works -- and guessing means either
|
||||
# a coroutine unpacked as a tuple, or the event loop blocked. None means
|
||||
# "do not patch"; apply.sh turns that into a skip, not a guess.
|
||||
half = dict(GOOD)
|
||||
half["plugins/cloud/snapshot.py"] = (
|
||||
'def create_snapshot(middleware, path, name="x"):\n return "s", "p"\n'
|
||||
)
|
||||
assert compat.async_flavour(loader(half)) is None
|
||||
|
||||
def test_an_unreadable_source_refuses_rather_than_guesses(self):
|
||||
broken = dict(GOOD)
|
||||
broken["plugins/cloud_backup/sync.py"] = Unreadable("HTTP 429")
|
||||
assert compat.async_flavour(loader(broken)) is None
|
||||
|
||||
def test_the_real_truenas_versions(self):
|
||||
# Pinning the actual fact this whole port exists for.
|
||||
assert compat.async_flavour(loader(GOOD)) is True
|
||||
|
||||
|
||||
class TestMiddlewareMethodsWeCall:
|
||||
"""The assumption class that was MISSING, and that hid a catastrophic break.
|
||||
|
||||
The manifest recorded the symbols the patch WRAPS. It said nothing about the
|
||||
middlewared methods the patch CALLS -- and TrueNAS 26 deleted
|
||||
plugins/zfs_/dataset.py and plugins/zfs_/snapshot.py outright, taking
|
||||
`zfs.dataset.query`, `zfs.snapshot.query` and `zfs.snapshot.delete` with them.
|
||||
|
||||
Nothing about the five cloud_backup files reveals that. The patch applied
|
||||
perfectly and every other check went green. The first backup would have failed --
|
||||
or, far worse, snapshotted fine and then failed to DELETE, orphaning one snapshot
|
||||
per descendant dataset (250 on a real pool) on every run, forever.
|
||||
"""
|
||||
|
||||
ZFS_SNAPSHOT = (
|
||||
"class ZFSSnapshot(CRUDService):\n"
|
||||
" class Config:\n"
|
||||
" namespace = 'zfs.snapshot'\n"
|
||||
" def query(self, filters, options):\n pass\n"
|
||||
" def delete(self, id_, options={}):\n pass\n"
|
||||
)
|
||||
ZFS_DATASET = (
|
||||
"class ZFSDataset(CRUDService):\n"
|
||||
" class Config:\n"
|
||||
" namespace = 'zfs.dataset'\n"
|
||||
" def query(self, filters, options):\n pass\n"
|
||||
)
|
||||
|
||||
def _tree(self, **over):
|
||||
files = dict(GOOD)
|
||||
files["plugins/zfs_/snapshot.py"] = self.ZFS_SNAPSHOT
|
||||
files["plugins/zfs_/dataset.py"] = self.ZFS_DATASET
|
||||
files.update(over)
|
||||
return files
|
||||
|
||||
def test_present_methods_are_ok(self):
|
||||
r = check_files(self._tree())
|
||||
assert r[NESTED]["ok"], r[NESTED]["problems"]
|
||||
|
||||
def test_a_deleted_plugin_file_is_broken(self):
|
||||
# Literally TrueNAS 26: plugins/zfs_/snapshot.py does not exist.
|
||||
r = check_files(self._tree(**{"plugins/zfs_/snapshot.py": None}))
|
||||
assert is_broken(r[NESTED])
|
||||
details = " ".join(p["detail"] for p in r[NESTED]["problems"])
|
||||
assert "zfs.snapshot.delete" in details
|
||||
|
||||
def test_a_renamed_namespace_is_broken(self):
|
||||
r = check_files(self._tree(**{
|
||||
"plugins/zfs_/snapshot.py": self.ZFS_SNAPSHOT.replace(
|
||||
"'zfs.snapshot'", "'zfs.resource.snapshot'"),
|
||||
}))
|
||||
assert is_broken(r[NESTED])
|
||||
|
||||
def test_the_CRUDService_do_prefix_is_accepted(self):
|
||||
# 24.10 and 25.04 declare `do_delete`; 25.10 renamed it to `delete`. BOTH
|
||||
# answer to zfs.snapshot.delete. Accepting only the literal name reported the
|
||||
# two older releases as broken -- a false BROKEN that would have switched off
|
||||
# nested snapshots on boxes where they work perfectly.
|
||||
r = check_files(self._tree(**{
|
||||
"plugins/zfs_/snapshot.py": self.ZFS_SNAPSHOT.replace(
|
||||
"def delete(", "def do_delete("),
|
||||
}))
|
||||
assert r[NESTED]["ok"], r[NESTED]["problems"]
|
||||
|
||||
def test_the_snapshot_delete_reason_names_the_orphan_risk(self):
|
||||
# If this ever regresses, whoever reads the bug report must understand that
|
||||
# it is not a cosmetic failure.
|
||||
r = check_files(self._tree(**{"plugins/zfs_/snapshot.py": None}))
|
||||
whys = " ".join(p["why"] for p in r[NESTED]["problems"])
|
||||
assert "orphan" in whys
|
||||
@@ -0,0 +1,133 @@
|
||||
"""The docs must not lie about themselves.
|
||||
|
||||
The README was 969 lines with the install instructions at line 517. Splitting it into
|
||||
docs/ fixed that and broke every cross-reference in the process -- which is the normal
|
||||
outcome of moving Markdown around, and exactly why this is a test rather than a
|
||||
careful afternoon.
|
||||
|
||||
A dead link in a recovery doc is worse than a dead link anywhere else: the person
|
||||
following it is, by definition, already having a bad day.
|
||||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
|
||||
import pytest
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
DOCS = os.path.join(ROOT, "docs")
|
||||
|
||||
LINK_RE = re.compile(r"\[([^\]]+)\]\(([^)]+)\)")
|
||||
HEADING_RE = re.compile(r"^#{1,6}\s+(.*)$", re.M)
|
||||
|
||||
|
||||
def markdown_files():
|
||||
files = [os.path.join(ROOT, "README.md"), os.path.join(ROOT, "CHANGELOG.md")]
|
||||
if os.path.isdir(DOCS):
|
||||
files += [os.path.join(DOCS, f) for f in sorted(os.listdir(DOCS))
|
||||
if f.endswith(".md")]
|
||||
return files
|
||||
|
||||
|
||||
def anchors(text):
|
||||
"""GitHub/Gitea slugs for every heading in `text`."""
|
||||
out = set()
|
||||
for h in HEADING_RE.findall(text):
|
||||
slug = re.sub(r"[^a-z0-9 -]", "", h.lower()).replace(" ", "-")
|
||||
out.add(slug)
|
||||
return out
|
||||
|
||||
|
||||
@pytest.mark.parametrize("path", markdown_files(), ids=os.path.basename)
|
||||
def test_every_internal_link_resolves(path):
|
||||
with open(path, encoding="utf-8") as fh:
|
||||
text = fh.read()
|
||||
here = anchors(text)
|
||||
base = os.path.dirname(path)
|
||||
|
||||
broken = []
|
||||
for label, target in LINK_RE.findall(text):
|
||||
if target.startswith(("http://", "https://", "mailto:")):
|
||||
continue
|
||||
rel, _, anchor = target.partition("#")
|
||||
|
||||
if not rel: # same-file anchor
|
||||
if anchor and anchor not in here:
|
||||
broken.append(f"[{label}](#{anchor}) — no such heading here")
|
||||
continue
|
||||
|
||||
dest = os.path.normpath(os.path.join(base, rel))
|
||||
if not os.path.exists(dest):
|
||||
broken.append(f"[{label}]({target}) — file does not exist")
|
||||
continue
|
||||
|
||||
if anchor and dest.endswith(".md"):
|
||||
with open(dest, encoding="utf-8") as fh:
|
||||
if anchor not in anchors(fh.read()):
|
||||
broken.append(f"[{label}]({target}) — no such heading there")
|
||||
|
||||
assert not broken, "broken links in {}:\n {}".format(
|
||||
os.path.basename(path), "\n ".join(broken)
|
||||
)
|
||||
|
||||
|
||||
class TestTheReadmeStaysAReadme:
|
||||
def test_install_is_near_the_top(self):
|
||||
# It was at line 517 of 969, under a wall of internals. Somebody deciding
|
||||
# whether to use this should not have to scroll past the boot sequence.
|
||||
with open(os.path.join(ROOT, "README.md"), encoding="utf-8") as fh:
|
||||
lines = fh.read().splitlines()
|
||||
install = next(i for i, ln in enumerate(lines, 1) if ln.startswith("## Install"))
|
||||
assert install < 40, f"## Install is at line {install}"
|
||||
|
||||
def test_the_readme_does_not_grow_back(self):
|
||||
with open(os.path.join(ROOT, "README.md"), encoding="utf-8") as fh:
|
||||
n = len(fh.read().splitlines())
|
||||
assert n < 300, (
|
||||
f"README is {n} lines. Detail belongs in docs/ — the README is what "
|
||||
f"someone reads before they trust this with their backups."
|
||||
)
|
||||
|
||||
def test_the_minimum_version_is_stated_before_the_install_command(self):
|
||||
with open(os.path.join(ROOT, "README.md"), encoding="utf-8") as fh:
|
||||
text = fh.read()
|
||||
assert "24.10" in text[:text.index("## Install")], (
|
||||
"the minimum TrueNAS version must be visible above the install steps"
|
||||
)
|
||||
|
||||
|
||||
class TestInstallDoesNotDirtyTheCheckout:
|
||||
"""install.sh chmod +x's scripts. If git records them as 100644, that chmod is a
|
||||
TRACKED MODIFICATION -- and update.sh refuses to run over a dirty tree.
|
||||
|
||||
So installing once permanently blocked updating, for every user, with a message
|
||||
telling them to `git checkout -- .` (which would just undo the exec bit and let
|
||||
the next install re-dirty it). Found on a real box that had been stuck on an old
|
||||
version for exactly this reason.
|
||||
|
||||
Every script install.sh makes executable must already be executable in git.
|
||||
"""
|
||||
|
||||
def test_every_chmodded_script_is_already_executable_in_git(self):
|
||||
import re
|
||||
import subprocess
|
||||
|
||||
with open(os.path.join(ROOT, "install.sh"), encoding="utf-8") as fh:
|
||||
m = re.search(r"^for _exe in (.+?); do", fh.read(), re.M)
|
||||
assert m, "could not find install.sh's chmod loop"
|
||||
scripts = m.group(1).split()
|
||||
|
||||
out = subprocess.run(
|
||||
["git", "ls-files", "-s", *scripts],
|
||||
cwd=ROOT, capture_output=True, text=True, check=True,
|
||||
).stdout
|
||||
|
||||
not_exec = [
|
||||
line.split("\t")[-1] for line in out.strip().splitlines()
|
||||
if not line.startswith("100755")
|
||||
]
|
||||
assert not not_exec, (
|
||||
"install.sh chmod +x's these, but git records them as non-executable — "
|
||||
"so installing dirties the checkout and update.sh then refuses to run:\n "
|
||||
+ "\n ".join(not_exec)
|
||||
)
|
||||
@@ -0,0 +1,264 @@
|
||||
"""Tests for the barrier: a stable release must have been a release candidate.
|
||||
|
||||
This exists because the repo cut twelve releases in one day, several of them
|
||||
fixing the release before -- and with the update alert live, every one of those
|
||||
interrupts every user. The gate makes that path impossible rather than impolite.
|
||||
|
||||
The provenance gate is the one thing here that can wrongly PASS in a way nobody
|
||||
notices (a wrongly-failing gate is loud; a wrongly-passing gate silently restores
|
||||
the old behaviour), so it gets tested against real git repositories, not mocks.
|
||||
"""
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "tools"))
|
||||
|
||||
from release_gate import ( # noqa: E402
|
||||
check_promotable,
|
||||
next_rc,
|
||||
rc_tags,
|
||||
)
|
||||
from release_notes import ( # noqa: E402
|
||||
base_version,
|
||||
is_prerelease,
|
||||
promote,
|
||||
unreleased_body,
|
||||
)
|
||||
|
||||
|
||||
# ── a real git repo, because the gate reads real tags ────────────────────────
|
||||
|
||||
class Repo:
|
||||
"""A throwaway git repo. The gate reads real tags, so the tests build real ones."""
|
||||
|
||||
def __init__(self, path):
|
||||
self.path = path
|
||||
|
||||
def __str__(self):
|
||||
return str(self.path)
|
||||
|
||||
def git(self, *args):
|
||||
return subprocess.run(
|
||||
["git", *args], cwd=self.path, capture_output=True, text=True, check=True,
|
||||
).stdout.strip()
|
||||
|
||||
def commit(self, msg):
|
||||
(self.path / "f").write_text(msg)
|
||||
self.git("add", "-A")
|
||||
self.git("commit", "-q", "-m", msg)
|
||||
return self.git("rev-parse", "HEAD")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def repo(tmp_path):
|
||||
d = tmp_path / "repo"
|
||||
d.mkdir()
|
||||
r = Repo(d)
|
||||
r.git("init", "-q", "-b", "main")
|
||||
r.git("config", "user.email", "t@example.com")
|
||||
r.git("config", "user.name", "t")
|
||||
r.commit("initial")
|
||||
return r
|
||||
|
||||
|
||||
class TestTheBarrierBeforeTheTagExists:
|
||||
"""release.sh calls the gate BEFORE creating the stable tag.
|
||||
|
||||
Every other test here tags first, and that is what let a fatal bug ship green:
|
||||
`check_promotable` began with `commit_for(vX.Y.Z)` and returned "does not exist",
|
||||
while release.sh's own guard refuses to run at all IF the tag exists. The two
|
||||
conditions were mutually exclusive, so `--promote` could never succeed -- the
|
||||
only way to cut a stable release was to hand-tag, bypassing every gate.
|
||||
|
||||
A gate that can only be satisfied after the thing it gates is not a gate.
|
||||
"""
|
||||
|
||||
def test_promote_is_allowed_when_head_was_a_candidate_and_the_tag_is_absent(self, repo):
|
||||
repo.git("tag", "v1.0.0-rc1") # rc on HEAD, no stable tag yet
|
||||
assert check_promotable("v1.0.0", cwd=str(repo)) == []
|
||||
|
||||
def test_promote_is_refused_when_head_was_never_a_candidate(self, repo):
|
||||
assert check_promotable("v1.0.0", cwd=str(repo))
|
||||
|
||||
def test_promote_is_refused_when_head_moved_past_the_candidate(self, repo):
|
||||
repo.git("tag", "v1.0.0-rc1")
|
||||
repo.commit("one more little fix") # HEAD is no longer the candidate
|
||||
problems = check_promotable("v1.0.0", cwd=str(repo))
|
||||
assert problems
|
||||
assert "no release candidate does" in problems[0]
|
||||
|
||||
|
||||
class TestTheBarrier:
|
||||
def test_a_tag_with_no_candidate_is_refused(self, repo):
|
||||
repo.git("tag", "v1.0.0")
|
||||
problems = check_promotable("v1.0.0", cwd=str(repo))
|
||||
assert problems
|
||||
assert "never a release candidate" in problems[0]
|
||||
|
||||
def test_a_tag_whose_candidate_is_on_the_same_commit_is_allowed(self, repo):
|
||||
repo.git("tag", "v1.0.0-rc1")
|
||||
repo.git("tag", "v1.0.0")
|
||||
assert check_promotable("v1.0.0", cwd=str(repo)) == []
|
||||
|
||||
def test_one_more_little_fix_after_the_rc_is_refused(self, repo):
|
||||
# THE case this whole mechanism exists for. The candidate passed, then a
|
||||
# "trivial" commit landed, and the stable tag ships code no candidate ever
|
||||
# tested. That is how v0.5.1 happened.
|
||||
repo.git("tag", "v1.0.0-rc1")
|
||||
repo.commit("just a tiny fix, surely fine")
|
||||
repo.git("tag", "v1.0.0")
|
||||
|
||||
problems = check_promotable("v1.0.0", cwd=str(repo))
|
||||
assert problems
|
||||
assert "no release candidate does" in problems[0]
|
||||
assert "v1.0.0-rc2" in problems[0], "must say how to fix it"
|
||||
|
||||
def test_an_rc_for_a_different_version_does_not_count(self, repo):
|
||||
repo.git("tag", "v0.9.0-rc1")
|
||||
repo.git("tag", "v1.0.0")
|
||||
problems = check_promotable("v1.0.0", cwd=str(repo))
|
||||
assert problems
|
||||
assert "never a release candidate" in problems[0]
|
||||
|
||||
def test_candidates_themselves_are_never_gated(self, repo):
|
||||
# Requiring an rc to have an rc would be a deadlock.
|
||||
repo.git("tag", "v1.0.0-rc1")
|
||||
assert check_promotable("v1.0.0-rc1", cwd=str(repo)) == []
|
||||
|
||||
def test_a_later_candidate_on_the_right_commit_rescues_it(self, repo):
|
||||
repo.git("tag", "v1.0.0-rc1")
|
||||
repo.commit("fix found during rc1")
|
||||
repo.git("tag", "v1.0.0-rc2") # re-cut on the fixed commit
|
||||
repo.git("tag", "v1.0.0")
|
||||
assert check_promotable("v1.0.0", cwd=str(repo)) == []
|
||||
|
||||
def test_a_tag_the_numbering_does_not_understand_is_not_a_candidate(self, repo):
|
||||
# `v1.0.0-rc*` also globs `v1.0.0-rc1-hotfix`, which _rc_number reads as 0.
|
||||
# The barrier must be satisfied only by something that really was a candidate.
|
||||
repo.git("tag", "v1.0.0-rc1-hotfix")
|
||||
repo.git("tag", "v1.0.0")
|
||||
problems = check_promotable("v1.0.0", cwd=str(repo))
|
||||
assert problems
|
||||
assert "never a release candidate" in problems[0]
|
||||
assert rc_tags("1.0.0", cwd=str(repo)) == []
|
||||
|
||||
|
||||
class TestRcNumbering:
|
||||
def test_first_candidate_is_rc1(self, repo):
|
||||
assert next_rc("1.0.0", cwd=str(repo)) == "v1.0.0-rc1"
|
||||
|
||||
def test_it_counts_up(self, repo):
|
||||
repo.git("tag", "v1.0.0-rc1")
|
||||
assert next_rc("1.0.0", cwd=str(repo)) == "v1.0.0-rc2"
|
||||
repo.git("tag", "v1.0.0-rc2")
|
||||
assert next_rc("1.0.0", cwd=str(repo)) == "v1.0.0-rc3"
|
||||
|
||||
def test_rc10_sorts_after_rc9_not_before(self, repo):
|
||||
# Lexicographic sorting would rank rc10 before rc9 and hand out a duplicate.
|
||||
for n in range(1, 11):
|
||||
repo.git("tag", f"v1.0.0-rc{n}")
|
||||
assert rc_tags("1.0.0", cwd=str(repo))[-1] == "v1.0.0-rc10"
|
||||
assert next_rc("1.0.0", cwd=str(repo)) == "v1.0.0-rc11"
|
||||
|
||||
def test_other_versions_do_not_leak_in(self, repo):
|
||||
repo.git("tag", "v0.9.0-rc7")
|
||||
assert next_rc("1.0.0", cwd=str(repo)) == "v1.0.0-rc1"
|
||||
|
||||
|
||||
# ── the content half of the gate ─────────────────────────────────────────────
|
||||
|
||||
class TestPrereleaseDetection:
|
||||
@pytest.mark.parametrize("tag", ["v1.2.3-rc1", "v1.2.3-rc10", "1.2.3-beta",
|
||||
"v1.2.3-alpha2", "V1.2.3-RC1"])
|
||||
def test_prereleases(self, tag):
|
||||
assert is_prerelease(tag)
|
||||
|
||||
@pytest.mark.parametrize("tag", ["v1.2.3", "1.2.3", "v0.0.1"])
|
||||
def test_stable(self, tag):
|
||||
assert not is_prerelease(tag)
|
||||
|
||||
def test_base_version_strips_the_suffix(self):
|
||||
assert base_version("v1.2.3-rc4") == "1.2.3"
|
||||
assert base_version("v1.2.3") == "1.2.3"
|
||||
|
||||
|
||||
class TestUnreleasedSection:
|
||||
def test_body_is_extracted(self):
|
||||
text = "# C\n\n## Unreleased\n\n### Fixed\n- a thing\n\n## v1.0.0 — 2026-01-01\n\n- old\n"
|
||||
assert "- a thing" in unreleased_body(text)
|
||||
assert "old" not in unreleased_body(text)
|
||||
|
||||
def test_empty_section_reads_as_empty(self):
|
||||
text = "# C\n\n## Unreleased\n\n## v1.0.0 — 2026-01-01\n\n- old\n"
|
||||
assert unreleased_body(text) == ""
|
||||
|
||||
def test_absent_section_reads_as_empty(self):
|
||||
assert unreleased_body("# C\n\n## v1.0.0 — 2026-01-01\n\n- old\n") == ""
|
||||
|
||||
def test_promote_renames_the_heading_and_keeps_the_body(self):
|
||||
text = "# C\n\n## Unreleased\n\n### Fixed\n- a thing\n\n## v1.0.0 — 2026-01-01\n"
|
||||
out = promote(text, "1.1.0", "2026-07-13")
|
||||
assert "## v1.1.0 — 2026-07-13" in out
|
||||
assert "## Unreleased" not in out
|
||||
assert "- a thing" in out
|
||||
assert "## v1.0.0 — 2026-01-01" in out, "older sections survive"
|
||||
|
||||
def test_promoting_nothing_is_refused(self):
|
||||
# A release with no content is a release nobody needed -- and it still
|
||||
# alerts every box.
|
||||
with pytest.raises(ValueError, match="nothing to release"):
|
||||
promote("# C\n\n## v1.0.0 — 2026-01-01\n", "1.1.0", "2026-07-13")
|
||||
|
||||
|
||||
class TestStrandedWorkBlocksAStableRelease:
|
||||
"""`check()` refuses a stable tag that leaves work under `## Unreleased`.
|
||||
|
||||
Either it is finished and belongs in the release, or the release is premature.
|
||||
"""
|
||||
|
||||
def _tree(self, tmp_path, changelog):
|
||||
from release_notes import VERSIONED_FILES
|
||||
for rel in VERSIONED_FILES:
|
||||
p = tmp_path / rel
|
||||
p.parent.mkdir(parents=True, exist_ok=True)
|
||||
marker = "__version__ = " if rel.endswith(".py") else "VERSION="
|
||||
p.write_text(f'{marker}"1.0.0"\n')
|
||||
(tmp_path / "CHANGELOG.md").write_text(changelog)
|
||||
return str(tmp_path)
|
||||
|
||||
def test_stranded_work_is_refused_for_a_stable_tag(self, tmp_path):
|
||||
from release_notes import check
|
||||
root = self._tree(tmp_path, (
|
||||
"# C\n\n## Unreleased\n\n### Fixed\n- not done yet\n\n"
|
||||
"## v1.0.0 — 2026-07-13\n\n### Added\n- the thing\n"
|
||||
))
|
||||
problems = check("v1.0.0", root=root)
|
||||
assert any("Unreleased" in p for p in problems)
|
||||
|
||||
def test_stranded_work_is_fine_for_a_candidate(self, tmp_path):
|
||||
# An rc may legitimately have more work queued behind it.
|
||||
from release_notes import check
|
||||
root = self._tree(tmp_path, (
|
||||
"# C\n\n## Unreleased\n\n### Fixed\n- later\n\n"
|
||||
"## v1.0.0 — 2026-07-13\n\n### Added\n- the thing\n"
|
||||
))
|
||||
assert check("v1.0.0-rc1", root=root) == []
|
||||
|
||||
def test_a_clean_stable_release_passes(self, tmp_path):
|
||||
from release_notes import check
|
||||
root = self._tree(tmp_path, (
|
||||
"# C\n\n## v1.0.0 — 2026-07-13\n\n### Added\n- the thing\n"
|
||||
))
|
||||
assert check("v1.0.0", root=root) == []
|
||||
|
||||
def test_a_candidate_checks_against_its_base_version(self, tmp_path):
|
||||
# The scripts say 1.0.0; the tag says v1.0.0-rc3. That must agree, not clash.
|
||||
from release_notes import check
|
||||
root = self._tree(tmp_path, (
|
||||
"# C\n\n## v1.0.0 — 2026-07-13\n\n### Added\n- the thing\n"
|
||||
))
|
||||
assert check("v1.0.0-rc3", root=root) == []
|
||||
+173
-2
@@ -9,6 +9,7 @@ across install.sh / uninstall.sh / recover.sh / apply.sh and nothing noticed.
|
||||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
@@ -21,6 +22,8 @@ from release_notes import ( # noqa: E402
|
||||
extract_notes,
|
||||
normalise,
|
||||
script_versions,
|
||||
significance,
|
||||
version_tuple,
|
||||
)
|
||||
|
||||
REPO = os.path.join(os.path.dirname(__file__), "..")
|
||||
@@ -106,9 +109,14 @@ class TestAgainstTheRealRepo:
|
||||
f"scripts say v{version} but the newest CHANGELOG entry is v{newest}"
|
||||
)
|
||||
|
||||
def test_check_passes_for_the_current_version(self):
|
||||
def test_the_repo_is_always_releasable_as_a_candidate(self):
|
||||
# Deliberately checked as an rc, not as a stable release. `main` carries
|
||||
# work under `## Unreleased` most of the time, and the stable gate refuses
|
||||
# that on purpose -- shipping with work stranded mid-section is how you get
|
||||
# a release that needs another release. So the invariant main must uphold is
|
||||
# the candidate one: versions agree, and the CHANGELOG section exists.
|
||||
version = next(iter({normalise(v) for v in script_versions(REPO).values()}))
|
||||
assert check(version, REPO) == []
|
||||
assert check(f"v{version}-rc1", REPO) == []
|
||||
|
||||
|
||||
class TestCheckCatchesMistakes:
|
||||
@@ -120,3 +128,166 @@ class TestCheckCatchesMistakes:
|
||||
def test_reports_a_missing_changelog_section(self):
|
||||
problems = check("v9.9.9", REPO)
|
||||
assert any("no section" in p for p in problems)
|
||||
|
||||
|
||||
class TestSignificance:
|
||||
"""Drives the TrueNAS update alert: what is worth bothering a human about.
|
||||
|
||||
The rule: a release whose CHANGELOG only has a "### Docs" section changed no
|
||||
code, and nobody should get an alert because a README was reworded.
|
||||
"""
|
||||
|
||||
TEXT = """\
|
||||
# Changelog
|
||||
|
||||
## v0.4.2 — 2026-07-13
|
||||
|
||||
### Docs
|
||||
|
||||
- reworded the README
|
||||
|
||||
## v0.4.1 — 2026-07-13
|
||||
|
||||
### Fixed
|
||||
|
||||
- a real bug
|
||||
|
||||
## v0.4.0 — 2026-07-13
|
||||
|
||||
### Added
|
||||
|
||||
- a feature
|
||||
|
||||
## v0.3.3 — 2026-07-13
|
||||
|
||||
### Security
|
||||
|
||||
- keep a password out of argv
|
||||
|
||||
## v0.3.2 — 2026-07-13
|
||||
|
||||
### Fixed
|
||||
|
||||
- something
|
||||
"""
|
||||
|
||||
def test_docs_only_release_does_not_alert(self):
|
||||
level, versions, _ = significance(self.TEXT, "0.4.1", "0.4.2")
|
||||
assert level == "docs"
|
||||
assert versions == ["0.4.2"]
|
||||
|
||||
def test_a_real_fix_alerts(self):
|
||||
level, _v, _h = significance(self.TEXT, "0.4.0", "0.4.1")
|
||||
assert level == "notable"
|
||||
|
||||
def test_security_in_range_escalates(self):
|
||||
level, _v, _h = significance(self.TEXT, "0.3.2", "0.3.3")
|
||||
assert level == "security"
|
||||
|
||||
def test_security_wins_even_when_the_newest_release_is_docs_only(self):
|
||||
# A docs-only v0.4.2 sitting on top of a security-fixing v0.3.3 must still
|
||||
# be reported as security — classify the whole span, not just the tip.
|
||||
level, versions, _ = significance(self.TEXT, "0.3.2", "0.4.2")
|
||||
assert level == "security"
|
||||
assert set(versions) == {"0.3.3", "0.4.0", "0.4.1", "0.4.2"}
|
||||
|
||||
def test_same_version_is_never_notable(self):
|
||||
level, versions, _ = significance(self.TEXT, "0.4.2", "0.4.2")
|
||||
assert level == "docs"
|
||||
assert versions == []
|
||||
|
||||
def test_range_is_exclusive_of_current_inclusive_of_latest(self):
|
||||
_l, versions, _h = significance(self.TEXT, "0.4.0", "0.4.2")
|
||||
assert "0.4.0" not in versions
|
||||
assert "0.4.2" in versions
|
||||
|
||||
def test_version_tuple_orders_correctly(self):
|
||||
assert version_tuple("v0.10.0") > version_tuple("v0.9.9")
|
||||
assert version_tuple("0.4.2") > version_tuple("0.4.1")
|
||||
# Pre-release suffixes are dropped, not ranked above the release.
|
||||
assert version_tuple("v0.5.0-rc1") == version_tuple("v0.5.0")
|
||||
|
||||
|
||||
class TestCandidateNotesResolveToTheBaseVersion:
|
||||
"""`notes v0.6.0-rc1` must return v0.6.0's section.
|
||||
|
||||
A candidate ships the same code as the release it is a candidate for, and the
|
||||
CHANGELOG only ever has the one section. Without this, the release workflow cut
|
||||
v0.6.0-rc1, passed every gate, and then died extracting the body -- so the tag
|
||||
existed but nothing was ever published. Caught in an rc, which is the entire
|
||||
point of having them.
|
||||
"""
|
||||
|
||||
CHANGELOG = "# C\n\n## v0.6.0 — 2026-07-13\n\n### Added\n- the thing\n\n## v0.5.1 — 2026-07-13\n\n- older\n"
|
||||
|
||||
def test_an_rc_resolves_to_its_base_version(self):
|
||||
body = extract_notes(self.CHANGELOG, "v0.6.0-rc1")
|
||||
assert "the thing" in body
|
||||
assert "older" not in body
|
||||
|
||||
def test_rc10_too(self):
|
||||
assert "the thing" in extract_notes(self.CHANGELOG, "v0.6.0-rc10")
|
||||
|
||||
def test_the_plain_version_still_works(self):
|
||||
assert "the thing" in extract_notes(self.CHANGELOG, "v0.6.0")
|
||||
|
||||
def test_a_genuinely_missing_section_still_raises(self):
|
||||
with pytest.raises(KeyError):
|
||||
extract_notes(self.CHANGELOG, "v9.9.9-rc1")
|
||||
|
||||
|
||||
class TestTheChangelogIsStructurallySound:
|
||||
"""The release body IS this file, so a mangled section ships to every user.
|
||||
|
||||
It has been mangled once: an edit matched the literal `## Unreleased` inside a
|
||||
backticked phrase in a prose bullet and spliced a whole new section into the middle
|
||||
of it, splitting the sentence in half.
|
||||
"""
|
||||
|
||||
def changelog(self):
|
||||
with open(os.path.join(REPO, "CHANGELOG.md"), encoding="utf-8") as fh:
|
||||
return fh.read()
|
||||
|
||||
def test_no_version_section_is_empty(self):
|
||||
text = self.changelog()
|
||||
for v in changelog_versions(text):
|
||||
assert extract_notes(text, v).strip(), f"v{v} has an empty section"
|
||||
|
||||
def test_versions_are_in_descending_order(self):
|
||||
from release_notes import version_tuple
|
||||
versions = changelog_versions(self.changelog())
|
||||
assert versions == sorted(versions, key=version_tuple, reverse=True), (
|
||||
"CHANGELOG versions are out of order — a section was spliced in wrong"
|
||||
)
|
||||
|
||||
def test_headings_are_at_the_start_of_a_line_and_not_inside_prose(self):
|
||||
# A `### Fixed` that ends up indented under a bullet is a section nobody sees.
|
||||
for i, line in enumerate(self.changelog().splitlines(), 1):
|
||||
if line.lstrip().startswith(("## ", "### ")) and line != line.lstrip():
|
||||
raise AssertionError(
|
||||
f"line {i}: heading is indented, so it is inside a list item "
|
||||
f"rather than being a section: {line!r}"
|
||||
)
|
||||
|
||||
def test_every_bullet_that_opens_a_bold_phrase_closes_it(self):
|
||||
# The splice cut `- **A stable release ... under \`## Unreleased` in half,
|
||||
# leaving an unterminated ** and a dangling sentence.
|
||||
#
|
||||
# A bullet is the `- ` line plus everything up to the next top-level bullet or
|
||||
# heading -- bold phrases routinely wrap across lines, so a per-line check
|
||||
# would flag every long bullet in the file.
|
||||
text = self.changelog()
|
||||
bullets = re.split(r"^(?=- |#{2,3} )", text, flags=re.M)
|
||||
bad = []
|
||||
for b in bullets:
|
||||
if not b.startswith("- "):
|
||||
continue
|
||||
# Code spans are not markup: `*args, **kwargs` is a literal, not a bold
|
||||
# phrase, and counting its ** would flag a perfectly well-formed bullet.
|
||||
prose = re.sub(r"`[^`]*`", "", b)
|
||||
if prose.count("**") % 2:
|
||||
bad.append(b.splitlines()[0][:70])
|
||||
assert not bad, (
|
||||
"unbalanced ** in a bullet — a section was probably spliced into the "
|
||||
"middle of it:\n " + "\n ".join(bad)
|
||||
)
|
||||
|
||||
+502
-56
@@ -12,7 +12,6 @@ Two rules are under test above all else:
|
||||
on EVERY run.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
import sys
|
||||
|
||||
@@ -199,12 +198,20 @@ class TestSnapshotTreeNames:
|
||||
|
||||
|
||||
class FakeMiddleware:
|
||||
"""middlewared as this module actually uses it: `call_sync`, from a thread.
|
||||
|
||||
The module is synchronous on purpose -- see the orchestration note in
|
||||
truecloud_nested.py. TrueNAS <= 25.10 reaches it through
|
||||
`await middleware.run_in_thread(...)` and TrueNAS 26 calls it directly, but the
|
||||
logic below the boundary is the same code either way, so it is tested once.
|
||||
"""
|
||||
|
||||
def __init__(self, snapshots=None):
|
||||
self.snapshots = list(snapshots or [])
|
||||
self.calls = []
|
||||
self.logger = None
|
||||
|
||||
async def call(self, method, *args):
|
||||
def call_sync(self, method, *args):
|
||||
self.calls.append((method, args))
|
||||
if method == "zfs.snapshot.query":
|
||||
return [{"name": n} for n in self.snapshots]
|
||||
@@ -222,8 +229,35 @@ class FakeMiddleware:
|
||||
return True
|
||||
raise AssertionError(f"unexpected call {method}")
|
||||
|
||||
async def run_in_thread(self, fn, *args):
|
||||
return fn(*args)
|
||||
|
||||
def stub_core(monkeypatch, tn, *, plan=None, order=None, plan_raises=None):
|
||||
"""Replace the blocking core (plan/apply/verify/teardown) with recorders.
|
||||
|
||||
stage_nested calls these directly now, so they are patched by NAME rather than
|
||||
intercepted at a `run_in_thread` boundary that no longer exists.
|
||||
"""
|
||||
def record(name, result):
|
||||
def fn(*args, **kwargs):
|
||||
if order is not None:
|
||||
order.append(name)
|
||||
if name == "plan_staging" and plan_raises is not None:
|
||||
raise plan_raises
|
||||
return result() if callable(result) else result
|
||||
fn.__name__ = name
|
||||
return fn
|
||||
|
||||
real_write = tn._write_sidecar
|
||||
|
||||
def write_sidecar(*args, **kwargs):
|
||||
if order is not None:
|
||||
order.append("_write_sidecar")
|
||||
return real_write(*args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(tn, "_write_sidecar", write_sidecar)
|
||||
monkeypatch.setattr(tn, "plan_staging", record("plan_staging", plan or ([], [])))
|
||||
monkeypatch.setattr(tn, "apply_plan", record("apply_plan", True))
|
||||
monkeypatch.setattr(tn, "verify_staged", record("verify_staged", True))
|
||||
monkeypatch.setattr(tn, "teardown", record("teardown", []))
|
||||
|
||||
|
||||
class TestDeleteSnapshotTree:
|
||||
@@ -231,20 +265,20 @@ class TestDeleteSnapshotTree:
|
||||
mw = FakeMiddleware([
|
||||
"Tap@snap", "Tap/apps@snap", "Tap/apps/lidarr@snap", "Tap@keepme",
|
||||
])
|
||||
asyncio.run(delete_snapshot_tree(mw, "Tap@snap"))
|
||||
delete_snapshot_tree(mw, "Tap@snap")
|
||||
assert mw.snapshots == ["Tap@keepme"]
|
||||
|
||||
def test_is_idempotent_when_stock_already_removed_the_parent(self):
|
||||
# Stock's finally can win the race once our mounts are released.
|
||||
mw = FakeMiddleware(["Tap/apps@snap", "Tap/apps/lidarr@snap"])
|
||||
asyncio.run(delete_snapshot_tree(mw, "Tap@snap"))
|
||||
delete_snapshot_tree(mw, "Tap@snap")
|
||||
assert mw.snapshots == []
|
||||
|
||||
def test_uses_a_single_recursive_delete_not_252_individual_ones(self):
|
||||
# 252 sequential deletes are slow AND not atomic: a run killed part-way
|
||||
# through leaves exactly the orphans this function exists to prevent.
|
||||
mw = FakeMiddleware(["Tap@snap", "Tap/apps@snap", "Tap/apps/lidarr@snap"])
|
||||
asyncio.run(delete_snapshot_tree(mw, "Tap@snap"))
|
||||
delete_snapshot_tree(mw, "Tap@snap")
|
||||
assert mw.snapshots == []
|
||||
deletes = [a for m, a in mw.calls if m == "zfs.snapshot.delete"]
|
||||
assert len(deletes) == 1, "should be ONE recursive call, not one per snapshot"
|
||||
@@ -255,20 +289,20 @@ class TestDeleteSnapshotTree:
|
||||
|
||||
def test_survives_recursive_and_query_failure_by_deleting_the_parent(self):
|
||||
class Broken(FakeMiddleware):
|
||||
async def call(self, method, *args):
|
||||
def call_sync(self, method, *args):
|
||||
if method == "zfs.snapshot.query":
|
||||
raise RuntimeError("boom")
|
||||
if method == "zfs.snapshot.delete" and len(args) > 1:
|
||||
raise RuntimeError("recursive delete unavailable")
|
||||
return await super().call(method, *args)
|
||||
return super().call_sync(method, *args)
|
||||
|
||||
mw = Broken(["Tap@snap"])
|
||||
asyncio.run(delete_snapshot_tree(mw, "Tap@snap"))
|
||||
delete_snapshot_tree(mw, "Tap@snap")
|
||||
assert mw.snapshots == []
|
||||
|
||||
def test_leaves_unrelated_snapshots_alone_when_the_tree_is_gone(self):
|
||||
mw = FakeMiddleware(["Tap@unrelated"])
|
||||
asyncio.run(delete_snapshot_tree(mw, "Tap@snap"))
|
||||
delete_snapshot_tree(mw, "Tap@snap")
|
||||
assert mw.snapshots == ["Tap@unrelated"]
|
||||
|
||||
|
||||
@@ -281,20 +315,13 @@ class TestStageNestedOrdering:
|
||||
|
||||
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
|
||||
order = []
|
||||
stub_core(monkeypatch, tn, order=order,
|
||||
plan=([("/src", str(tmp_path / "cloud_backup-5"))], []))
|
||||
|
||||
class Recorder(FakeMiddleware):
|
||||
async def run_in_thread(self, fn, *args):
|
||||
order.append(fn.__name__)
|
||||
if fn.__name__ == "plan_staging":
|
||||
return ([("/src", str(tmp_path / "cloud_backup-5"))], [])
|
||||
if fn.__name__ in ("apply_plan", "verify_staged", "teardown"):
|
||||
return [] if fn.__name__ == "teardown" else True
|
||||
return fn(*args)
|
||||
|
||||
asyncio.run(tn.stage_nested(
|
||||
Recorder(), "/mnt/Tap", "Tap@snap", "Tap", "/mnt/Tap",
|
||||
tn.stage_nested(
|
||||
FakeMiddleware(), "/mnt/Tap", "Tap@snap", "Tap", "/mnt/Tap",
|
||||
"cloud_backup-5", DATASETS,
|
||||
))
|
||||
)
|
||||
|
||||
assert order.index("_write_sidecar") < order.index("apply_plan")
|
||||
|
||||
@@ -312,50 +339,41 @@ class TestStageNestedOrdering:
|
||||
fh.write("Tap@old-crashed-run")
|
||||
|
||||
mw = FakeMiddleware(["Tap@old-crashed-run", "Tap/apps@old-crashed-run"])
|
||||
stub_core(monkeypatch, tn, plan=([("/src", root)], []))
|
||||
|
||||
class Stub(FakeMiddleware):
|
||||
def __init__(self, inner):
|
||||
super().__init__()
|
||||
self.inner = inner
|
||||
|
||||
async def call(self, method, *args):
|
||||
return await self.inner.call(method, *args)
|
||||
|
||||
async def run_in_thread(self, fn, *args):
|
||||
if fn.__name__ == "plan_staging":
|
||||
return ([("/src", root)], [])
|
||||
if fn.__name__ == "teardown":
|
||||
return []
|
||||
if fn.__name__ in ("apply_plan", "verify_staged"):
|
||||
return True
|
||||
return fn(*args)
|
||||
|
||||
asyncio.run(tn.stage_nested(
|
||||
Stub(mw), "/mnt/Tap", "Tap@new", "Tap", "/mnt/Tap",
|
||||
tn.stage_nested(
|
||||
mw, "/mnt/Tap", "Tap@new", "Tap", "/mnt/Tap",
|
||||
"cloud_backup-5", DATASETS,
|
||||
))
|
||||
)
|
||||
|
||||
assert mw.snapshots == [], "the crashed run's snapshot tree must be reclaimed"
|
||||
|
||||
def test_sidecar_is_removed_when_staging_fails(self, tmp_path, monkeypatch):
|
||||
def test_sidecar_is_KEPT_when_staging_fails(self, tmp_path, monkeypatch):
|
||||
# The caller sweeps the snapshot tree on the way out, and anything still busy
|
||||
# SURVIVES that sweep -- with the sidecar as its only record. Removing the
|
||||
# sidecar here would orphan those snapshots permanently.
|
||||
#
|
||||
# The asymmetry is the point: a sidecar left behind when the tree is already
|
||||
# gone costs one no-op delete on the next run; a sidecar removed while the tree
|
||||
# still exists is unrecoverable.
|
||||
import truecloud_nested as tn
|
||||
|
||||
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
|
||||
root = tn.staging_root_for("cloud_backup-5")
|
||||
|
||||
class Failing(FakeMiddleware):
|
||||
async def run_in_thread(self, fn, *args):
|
||||
if fn.__name__ == "plan_staging":
|
||||
raise StagingError("boom")
|
||||
return fn(*args)
|
||||
stub_core(monkeypatch, tn, plan_raises=StagingError("boom"))
|
||||
|
||||
with pytest.raises(StagingError):
|
||||
asyncio.run(tn.stage_nested(
|
||||
Failing(), "/mnt/Tap", "Tap@snap", "Tap", "/mnt/Tap",
|
||||
tn.stage_nested(
|
||||
FakeMiddleware(), "/mnt/Tap", "Tap@snap", "Tap", "/mnt/Tap",
|
||||
"cloud_backup-5", DATASETS,
|
||||
))
|
||||
)
|
||||
|
||||
assert not os.path.exists(sidecar_for(root))
|
||||
assert os.path.exists(sidecar_for(root)), (
|
||||
"sidecar removed on staging failure — any snapshot the caller's sweep "
|
||||
"cannot delete is now orphaned forever"
|
||||
)
|
||||
with open(sidecar_for(root), encoding="utf-8") as fh:
|
||||
assert fh.read().strip() == "Tap@snap"
|
||||
|
||||
|
||||
class TestCleanupTask:
|
||||
@@ -374,7 +392,7 @@ class TestCleanupTask:
|
||||
mw = FakeMiddleware(["Tap@snap", "Tap/apps@snap"])
|
||||
monkeypatch.setattr(tn, "teardown", lambda *_a, **_k: [])
|
||||
|
||||
asyncio.run(cleanup_task(mw, "cloud_backup-5"))
|
||||
cleanup_task(mw, "cloud_backup-5")
|
||||
|
||||
assert mw.snapshots == []
|
||||
assert not os.path.exists(sidecar_for(root))
|
||||
@@ -384,7 +402,7 @@ class TestCleanupTask:
|
||||
|
||||
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path / "nope"))
|
||||
mw = FakeMiddleware(["Tap@snap"])
|
||||
asyncio.run(cleanup_task(mw, "cloud_backup-5"))
|
||||
cleanup_task(mw, "cloud_backup-5")
|
||||
assert mw.calls == []
|
||||
assert mw.snapshots == ["Tap@snap"]
|
||||
|
||||
@@ -603,3 +621,431 @@ class TestStagingRootFor:
|
||||
# os.path.join(BASE, "..") normalises to /run — teardown would rmdir it.
|
||||
root = staging_root_for(name)
|
||||
assert os.path.normpath(root).startswith("/run/truecloud-nested/")
|
||||
|
||||
|
||||
class TestZfsAutomountKeepsSnapshotsBusy:
|
||||
""""dataset is busy" is EXPECTED, TRANSIENT, and used to orphan snapshots forever.
|
||||
|
||||
Reading `<dataset>/.zfs/snapshot/<snap>/` makes ZFS **automount** that snapshot,
|
||||
and it stays mounted for zfs_expire_snapshot seconds (300 by default) after the
|
||||
last access. teardown() unmounts OUR bind mounts, but not the automount underneath
|
||||
-- so `zfs destroy` refuses with EBUSY for everything restic read recently.
|
||||
|
||||
Observed on a real 256-snapshot tree: 253 swept cleanly, and the 3 datasets restic
|
||||
had touched last failed with "dataset is busy". cleanup_task then removed the
|
||||
sidecar anyway, so nothing would ever reclaim them. A few snapshots leaked per run,
|
||||
forever.
|
||||
"""
|
||||
|
||||
MOUNTS = (
|
||||
"tmpfs /run tmpfs rw 0 0\n"
|
||||
"Tap/apps/prometheus /mnt/Tap/apps/prometheus/.zfs/snapshot/snap1 zfs ro 0 0\n"
|
||||
"Tap/apps/standing/data /mnt/Tap/apps/standing/data/.zfs/snapshot/snap1 zfs ro 0 0\n"
|
||||
"Tap /mnt/Tap/.zfs/snapshot/snap1 zfs ro 0 0\n"
|
||||
"Tap/other /mnt/Tap/other/.zfs/snapshot/OTHER zfs ro 0 0\n"
|
||||
)
|
||||
|
||||
def _mounts_file(self, tmp_path):
|
||||
p = tmp_path / "mounts"
|
||||
p.write_text(self.MOUNTS)
|
||||
return str(p)
|
||||
|
||||
def test_it_finds_the_automounts_for_this_snapshot_only(self, tmp_path):
|
||||
import truecloud_nested as tn
|
||||
found = tn.snapdir_automounts("snap1", mounts_file=self._mounts_file(tmp_path))
|
||||
assert "/mnt/Tap/other/.zfs/snapshot/OTHER" not in found
|
||||
assert len(found) == 3
|
||||
|
||||
def test_deepest_first(self, tmp_path):
|
||||
# A child's automount must be released before its parent's.
|
||||
import truecloud_nested as tn
|
||||
found = tn.snapdir_automounts("snap1", mounts_file=self._mounts_file(tmp_path))
|
||||
assert found[-1] == "/mnt/Tap/.zfs/snapshot/snap1"
|
||||
|
||||
def test_release_snapdirs_unmounts_them(self, tmp_path):
|
||||
import truecloud_nested as tn
|
||||
called = []
|
||||
|
||||
class R:
|
||||
returncode = 0
|
||||
stderr = ""
|
||||
|
||||
def runner(cmd):
|
||||
called.append(cmd)
|
||||
return R()
|
||||
|
||||
errs = tn.release_snapdirs("snap1", runner=runner,
|
||||
mounts_file=self._mounts_file(tmp_path))
|
||||
assert errs == []
|
||||
assert all(c[0] == "umount" for c in called)
|
||||
assert len(called) == 3
|
||||
|
||||
|
||||
class BusyMiddleware(FakeMiddleware):
|
||||
"""Deletes fail with EBUSY until `busy_until_attempt` passes -- like a ZFS
|
||||
automount expiring."""
|
||||
|
||||
def __init__(self, snapshots, busy, busy_for=2):
|
||||
super().__init__(snapshots)
|
||||
self.busy = set(busy)
|
||||
self.busy_for = busy_for
|
||||
self.attempts = 0
|
||||
|
||||
def call_sync(self, method, *args):
|
||||
if method == "zfs.snapshot.delete":
|
||||
name = args[0]
|
||||
opts = args[1] if len(args) > 1 else {}
|
||||
if opts.get("recursive"):
|
||||
raise RuntimeError("cannot destroy snapshot: dataset is busy")
|
||||
if name in self.busy:
|
||||
self.attempts += 1
|
||||
if self.attempts <= self.busy_for * len(self.busy):
|
||||
raise RuntimeError(f"cannot destroy '{name}': dataset is busy")
|
||||
return super().call_sync(method, *args)
|
||||
|
||||
|
||||
class TestDeleteRetriesAndReportsSurvivors:
|
||||
def test_a_transient_busy_is_retried_and_wins(self, monkeypatch):
|
||||
import truecloud_nested as tn
|
||||
monkeypatch.setattr(tn, "release_snapdirs", lambda *a, **k: [])
|
||||
|
||||
mw = BusyMiddleware(
|
||||
["Tap@snap", "Tap/apps@snap", "Tap/apps/prometheus@snap"],
|
||||
busy=["Tap/apps/prometheus@snap"], busy_for=1,
|
||||
)
|
||||
survivors = tn.delete_snapshot_tree(mw, "Tap@snap", sleep=lambda _s: None)
|
||||
assert survivors == []
|
||||
assert mw.snapshots == []
|
||||
|
||||
def test_a_permanently_busy_snapshot_is_REPORTED_not_swallowed(self, monkeypatch):
|
||||
import truecloud_nested as tn
|
||||
monkeypatch.setattr(tn, "release_snapdirs", lambda *a, **k: [])
|
||||
|
||||
mw = BusyMiddleware(
|
||||
["Tap@snap", "Tap/apps/prometheus@snap"],
|
||||
busy=["Tap/apps/prometheus@snap"], busy_for=99,
|
||||
)
|
||||
survivors = tn.delete_snapshot_tree(mw, "Tap@snap", sleep=lambda _s: None)
|
||||
assert survivors == ["Tap/apps/prometheus@snap"]
|
||||
assert mw.snapshots == ["Tap/apps/prometheus@snap"]
|
||||
|
||||
def test_the_automounts_are_released_before_deleting(self, monkeypatch):
|
||||
import truecloud_nested as tn
|
||||
order = []
|
||||
monkeypatch.setattr(tn, "release_snapdirs",
|
||||
lambda name, **k: order.append(("release", name)) or [])
|
||||
mw = FakeMiddleware(["Tap@snap"])
|
||||
real = mw.call_sync
|
||||
|
||||
def spy(method, *args):
|
||||
order.append((method, args[0] if args else None))
|
||||
return real(method, *args)
|
||||
|
||||
mw.call_sync = spy
|
||||
tn.delete_snapshot_tree(mw, "Tap@snap", sleep=lambda _s: None)
|
||||
assert order[0] == ("release", "snap"), order
|
||||
|
||||
|
||||
class TestSidecarSurvivesAnIncompleteSweep:
|
||||
def test_the_sidecar_is_KEPT_when_snapshots_could_not_be_deleted(
|
||||
self, tmp_path, monkeypatch
|
||||
):
|
||||
# It is the ONLY record those snapshots exist. Removing it orphans them
|
||||
# permanently -- which is exactly what happened on the real box.
|
||||
import truecloud_nested as tn
|
||||
|
||||
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
|
||||
monkeypatch.setattr(tn, "release_snapdirs", lambda *a, **k: [])
|
||||
root = tn.staging_root_for("cloud_backup-5")
|
||||
os.makedirs(root, exist_ok=True)
|
||||
with open(sidecar_for(root), "w", encoding="utf-8") as fh:
|
||||
fh.write("Tap@snap")
|
||||
|
||||
mw = BusyMiddleware(["Tap@snap", "Tap/apps/prometheus@snap"],
|
||||
busy=["Tap/apps/prometheus@snap"], busy_for=99)
|
||||
monkeypatch.setattr(tn, "delete_snapshot_tree",
|
||||
lambda m, s, logger=None: ["Tap/apps/prometheus@snap"])
|
||||
|
||||
tn.cleanup_task(mw, "cloud_backup-5")
|
||||
assert os.path.exists(sidecar_for(root)), (
|
||||
"sidecar removed despite survivors — they are now orphaned forever"
|
||||
)
|
||||
|
||||
def test_the_sidecar_is_removed_on_a_clean_sweep(self, tmp_path, monkeypatch):
|
||||
import truecloud_nested as tn
|
||||
|
||||
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
|
||||
root = tn.staging_root_for("cloud_backup-5")
|
||||
os.makedirs(root, exist_ok=True)
|
||||
with open(sidecar_for(root), "w", encoding="utf-8") as fh:
|
||||
fh.write("Tap@snap")
|
||||
|
||||
monkeypatch.setattr(tn, "delete_snapshot_tree", lambda m, s, logger=None: [])
|
||||
tn.cleanup_task(FakeMiddleware(), "cloud_backup-5")
|
||||
assert not os.path.exists(sidecar_for(root))
|
||||
|
||||
|
||||
class TestTheSidecarCarriesEveryPendingTree:
|
||||
"""The sidecar holds a LIST, and that is a bug fix, not a generalisation.
|
||||
|
||||
It used to hold ONE snapshot. So a run that reclaimed an older tree, FAILED to
|
||||
finish reclaiming it, and then recorded its own snapshot would **overwrite the only
|
||||
record of the survivor** — orphaning it permanently, via the exact code written to
|
||||
prevent orphans.
|
||||
|
||||
Observed live: a snapshot survived one run; the next run's reclaim also failed
|
||||
(ZFS's 300s automount window had not elapsed, because the two runs were minutes
|
||||
apart); the record was overwritten; the snapshot was orphaned for good.
|
||||
"""
|
||||
|
||||
def test_round_trips_a_list(self, tmp_path):
|
||||
import truecloud_nested as tn
|
||||
root = str(tmp_path / "cloud_backup-5")
|
||||
tn._write_sidecar(root, ["Tap@a", "Tap@b"])
|
||||
assert tn._read_sidecar(root) == ["Tap@a", "Tap@b"]
|
||||
|
||||
def test_reads_the_old_single_line_format(self, tmp_path):
|
||||
# Boxes upgrading from an older version have a one-line sidecar on disk.
|
||||
import truecloud_nested as tn
|
||||
root = str(tmp_path / "cloud_backup-5")
|
||||
os.makedirs(os.path.dirname(sidecar_for(root)), exist_ok=True)
|
||||
with open(sidecar_for(root), "w", encoding="utf-8") as fh:
|
||||
fh.write("Tap@legacy")
|
||||
assert tn._read_sidecar(root) == ["Tap@legacy"]
|
||||
|
||||
def test_a_failed_reclaim_is_carried_forward_not_overwritten(
|
||||
self, tmp_path, monkeypatch
|
||||
):
|
||||
# THE bug. stage_nested reclaims an old tree, cannot finish, then records its
|
||||
# own snapshot -- the survivor must still be in the sidecar afterwards.
|
||||
import truecloud_nested as tn
|
||||
|
||||
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
|
||||
root = tn.staging_root_for("cloud_backup-5")
|
||||
os.makedirs(os.path.dirname(root), exist_ok=True)
|
||||
tn._write_sidecar(root, ["Tap@old"])
|
||||
|
||||
# The reclaim of Tap@old leaves one snapshot behind (still busy).
|
||||
monkeypatch.setattr(
|
||||
tn, "delete_snapshot_tree",
|
||||
lambda m, s, logger=None: ["Tap/apps/x@old"] if s == "Tap@old" else [],
|
||||
)
|
||||
stub_core(monkeypatch, tn, plan=([("/src", root)], []))
|
||||
|
||||
tn.stage_nested(FakeMiddleware(), "/mnt/Tap", "Tap@new", "Tap", "/mnt/Tap",
|
||||
"cloud_backup-5", DATASETS)
|
||||
|
||||
recorded = tn._read_sidecar(root)
|
||||
assert "Tap/apps/x@old" in recorded, (
|
||||
"the failed reclaim's survivor was dropped — orphaned forever"
|
||||
)
|
||||
assert "Tap@new" in recorded, "our own snapshot must also be recorded"
|
||||
|
||||
def test_cleanup_sweeps_every_pending_tree_and_records_only_survivors(
|
||||
self, tmp_path, monkeypatch
|
||||
):
|
||||
import truecloud_nested as tn
|
||||
|
||||
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
|
||||
root = tn.staging_root_for("cloud_backup-5")
|
||||
os.makedirs(root, exist_ok=True)
|
||||
tn._write_sidecar(root, ["Tap@old", "Tap@new"])
|
||||
|
||||
swept = []
|
||||
|
||||
def fake_delete(m, s, logger=None):
|
||||
swept.append(s)
|
||||
return ["Tap/apps/x@new"] if s == "Tap@new" else []
|
||||
|
||||
monkeypatch.setattr(tn, "delete_snapshot_tree", fake_delete)
|
||||
tn.cleanup_task(FakeMiddleware(), "cloud_backup-5")
|
||||
|
||||
assert swept == ["Tap@old", "Tap@new"], "both pending trees must be swept"
|
||||
# Only the SURVIVOR is written back -- re-recording Tap@old would make every
|
||||
# future run re-sweep a tree that is already gone.
|
||||
assert tn._read_sidecar(root) == ["Tap/apps/x@new"]
|
||||
|
||||
def test_a_fully_clean_sweep_removes_the_sidecar(self, tmp_path, monkeypatch):
|
||||
import truecloud_nested as tn
|
||||
|
||||
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
|
||||
root = tn.staging_root_for("cloud_backup-5")
|
||||
os.makedirs(root, exist_ok=True)
|
||||
tn._write_sidecar(root, ["Tap@a", "Tap@b"])
|
||||
monkeypatch.setattr(tn, "delete_snapshot_tree", lambda m, s, logger=None: [])
|
||||
|
||||
tn.cleanup_task(FakeMiddleware(), "cloud_backup-5")
|
||||
assert not os.path.exists(sidecar_for(root))
|
||||
|
||||
def test_cleanup_all_reports_each_pending_snapshot_on_its_own_line(self, tmp_path):
|
||||
# It formats them for a human during uninstall. A list rendered into an
|
||||
# f-string would print "['Tap@a', 'Tap@b']" at them.
|
||||
import truecloud_nested as tn
|
||||
root = str(tmp_path / "cloud_backup-5")
|
||||
tn._write_sidecar(root, ["Tap@a", "Tap@b"])
|
||||
|
||||
lines, _errors = tn.cleanup_all(
|
||||
base=str(tmp_path),
|
||||
glob_fn=lambda _p: [sidecar_for(root)],
|
||||
mounts_file=os.devnull,
|
||||
)
|
||||
notes = [ln for ln in lines if "left snapshot" in ln]
|
||||
assert len(notes) == 2
|
||||
assert "'Tap@a'" in notes[0] and "'Tap@b'" in notes[1]
|
||||
assert "[" not in "".join(notes)
|
||||
|
||||
|
||||
class TestGarbageCollectorSelection:
|
||||
"""`stale_snapshot_names` DELETES DATA on a name match.
|
||||
|
||||
A name match is a weaker claim than a recorded fact, so every way it could be wrong
|
||||
is a test. It exists because the sidecar — which IS a recorded fact — lives in /run,
|
||||
which is tmpfs: a reboot mid-backup destroys it and orphans a 250-snapshot tree with
|
||||
nothing left pointing at it. This is the only thing that would ever find those.
|
||||
"""
|
||||
|
||||
import datetime as _dt
|
||||
NOW = _dt.datetime(2026, 7, 14, 12, 0, 0, tzinfo=_dt.UTC)
|
||||
CURRENT = "Tap@cloud_backup-5-20260714115900" # 1 minute ago
|
||||
OLD = "Tap/apps/x@cloud_backup-5-20260713030000" # ~33 hours ago
|
||||
|
||||
def collect(self, names, **kw):
|
||||
import truecloud_nested as tn
|
||||
return tn.stale_snapshot_names(
|
||||
"cloud_backup-5", self.CURRENT, names, self.NOW, **kw
|
||||
)
|
||||
|
||||
def test_it_collects_our_own_leftovers(self):
|
||||
assert self.collect([self.OLD]) == [self.OLD]
|
||||
|
||||
def test_it_NEVER_touches_the_current_run(self):
|
||||
# Both the parent and its children share the current snapname.
|
||||
names = [self.CURRENT, "Tap/apps/x@cloud_backup-5-20260714115900"]
|
||||
assert self.collect(names) == []
|
||||
|
||||
def test_it_NEVER_touches_a_periodic_snapshot(self):
|
||||
assert self.collect(["Tap/apps/x@auto-2026-07-13_03-00"]) == []
|
||||
|
||||
def test_it_NEVER_touches_a_human_made_snapshot(self):
|
||||
assert self.collect(["Tap@before-i-broke-everything"]) == []
|
||||
|
||||
def test_it_NEVER_touches_another_TASK(self):
|
||||
# cloud_backup-5 must not match cloud_backup-50. This is why the prefix
|
||||
# carries the trailing dash.
|
||||
assert self.collect(["Tap/apps/x@cloud_backup-50-20260713030000"]) == []
|
||||
assert self.collect(["Tap/apps/x@cloud_backup-7-20260713030000"]) == []
|
||||
|
||||
def test_it_NEVER_touches_a_one_time_backup(self):
|
||||
assert self.collect(["Tap@cloud_backup-onetime-20260713030000"]) == []
|
||||
|
||||
def test_it_NEVER_touches_a_snapshot_that_is_MOUNTED(self):
|
||||
# An in-flight run pins its own snapshots. This — not the age heuristic — is
|
||||
# what actually protects a concurrent backup.
|
||||
assert self.collect([self.OLD], in_use={self.OLD}) == []
|
||||
|
||||
def test_it_NEVER_touches_a_snapshot_younger_than_the_minimum_age(self):
|
||||
# Covers the seconds-long window between `zfs snapshot -r` and the mounts
|
||||
# appearing, when a live run's snapshots look exactly like garbage.
|
||||
young = "Tap/apps/x@cloud_backup-5-20260714113000" # 30 minutes ago
|
||||
assert self.collect([young]) == []
|
||||
assert self.collect([young], min_age=60) == [young]
|
||||
|
||||
def test_a_name_it_cannot_parse_is_left_alone(self):
|
||||
assert self.collect(["Tap@cloud_backup-5-not-a-timestamp"]) == []
|
||||
assert self.collect(["Tap@cloud_backup-5-"]) == []
|
||||
|
||||
def test_a_realistic_mixed_pool(self):
|
||||
names = [
|
||||
self.CURRENT, # ours, running
|
||||
"Tap/apps/x@cloud_backup-5-20260714115900", # ours, running (child)
|
||||
self.OLD, # ours, orphaned <-
|
||||
"Tap/apps/y@cloud_backup-5-20260712030000", # ours, orphaned <-
|
||||
"Tap/apps/x@auto-2026-07-13_03-00", # periodic
|
||||
"Tap/apps/x@cloud_backup-7-20260713030000", # another task
|
||||
"Tap@manual-keepme", # human
|
||||
]
|
||||
assert sorted(self.collect(names)) == sorted(
|
||||
[self.OLD, "Tap/apps/y@cloud_backup-5-20260712030000"]
|
||||
)
|
||||
|
||||
|
||||
class TestMountedSnapshots:
|
||||
def test_it_reads_snapshot_names_out_of_the_mount_table(self, tmp_path):
|
||||
import truecloud_nested as tn
|
||||
mounts = tmp_path / "mounts"
|
||||
mounts.write_text(
|
||||
"tmpfs /run tmpfs rw 0 0\n"
|
||||
"Tap/apps/x@snap1 /run/truecloud-nested/t/apps/x zfs ro 0 0\n"
|
||||
"Tap/apps/y@snap1 /mnt/Tap/apps/y/.zfs/snapshot/snap1 zfs ro 0 0\n"
|
||||
"Tap/live /mnt/Tap/live zfs rw 0 0\n"
|
||||
)
|
||||
live = tn.mounted_snapshots(str(mounts))
|
||||
assert live == {"Tap/apps/x@snap1", "Tap/apps/y@snap1"}
|
||||
assert "Tap/live" not in live # a live dataset is not a snapshot
|
||||
|
||||
|
||||
class TestGarbageCollectorExecution:
|
||||
def test_it_deletes_the_stale_ones_and_nothing_else(self, monkeypatch, tmp_path):
|
||||
import datetime as dt
|
||||
import truecloud_nested as tn
|
||||
|
||||
mounts = tmp_path / "mounts"
|
||||
mounts.write_text("")
|
||||
now = dt.datetime(2026, 7, 14, 12, 0, 0, tzinfo=dt.UTC)
|
||||
|
||||
mw = FakeMiddleware([
|
||||
"Tap@cloud_backup-5-20260714115900", # current run
|
||||
"Tap/apps/x@cloud_backup-5-20260713030000", # orphan <-
|
||||
"Tap/apps/x@auto-2026-07-13_03-00", # periodic
|
||||
"Tap/apps/x@cloud_backup-7-20260713030000", # other task
|
||||
])
|
||||
remaining = tn.gc_stale_snapshots(
|
||||
mw, "cloud_backup-5", "Tap@cloud_backup-5-20260714115900",
|
||||
now=now, mounts_file=str(mounts),
|
||||
)
|
||||
assert remaining == []
|
||||
assert mw.snapshots == [
|
||||
"Tap@cloud_backup-5-20260714115900",
|
||||
"Tap/apps/x@auto-2026-07-13_03-00",
|
||||
"Tap/apps/x@cloud_backup-7-20260713030000",
|
||||
]
|
||||
|
||||
def test_a_busy_orphan_is_reported_not_swallowed(self, monkeypatch, tmp_path):
|
||||
import datetime as dt
|
||||
import truecloud_nested as tn
|
||||
|
||||
mounts = tmp_path / "mounts"
|
||||
mounts.write_text("")
|
||||
now = dt.datetime(2026, 7, 14, 12, 0, 0, tzinfo=dt.UTC)
|
||||
orphan = "Tap/apps/x@cloud_backup-5-20260713030000"
|
||||
|
||||
mw = BusyMiddleware(
|
||||
["Tap@cloud_backup-5-20260714115900", orphan],
|
||||
busy=[orphan], busy_for=99,
|
||||
)
|
||||
remaining = tn.gc_stale_snapshots(
|
||||
mw, "cloud_backup-5", "Tap@cloud_backup-5-20260714115900",
|
||||
now=now, mounts_file=str(mounts),
|
||||
)
|
||||
assert remaining == [orphan]
|
||||
|
||||
def test_it_collects_NOTHING_when_the_query_fails(self, tmp_path):
|
||||
# Cannot enumerate => cannot know what is ours => delete nothing.
|
||||
import datetime as dt
|
||||
import truecloud_nested as tn
|
||||
|
||||
mounts = tmp_path / "mounts"
|
||||
mounts.write_text("")
|
||||
|
||||
class Broken(FakeMiddleware):
|
||||
def call_sync(self, method, *args):
|
||||
if method == "zfs.snapshot.query":
|
||||
raise RuntimeError("middleware is having a day")
|
||||
return super().call_sync(method, *args)
|
||||
|
||||
assert tn.gc_stale_snapshots(
|
||||
Broken(["Tap/apps/x@cloud_backup-5-20260713030000"]),
|
||||
"cloud_backup-5", "Tap@cloud_backup-5-20260714115900",
|
||||
now=dt.datetime(2026, 7, 14, 12, 0, 0, tzinfo=dt.UTC),
|
||||
mounts_file=str(mounts),
|
||||
) == []
|
||||
|
||||
@@ -0,0 +1,107 @@
|
||||
"""Guards on the CI workflows themselves.
|
||||
|
||||
The workflows run on TWO forges -- Gitea (canonical) and GitHub (mirror), because
|
||||
Gitea reads .github/workflows too -- and they hold tokens. A mistake here is not a
|
||||
failed build, it is a bug report nobody files or a command nobody meant to run.
|
||||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
|
||||
import pytest
|
||||
|
||||
WORKFLOWS = os.path.join(os.path.dirname(__file__), "..", ".github", "workflows")
|
||||
|
||||
|
||||
def workflow_files():
|
||||
return [
|
||||
os.path.join(WORKFLOWS, f)
|
||||
for f in sorted(os.listdir(WORKFLOWS))
|
||||
if f.endswith((".yml", ".yaml"))
|
||||
]
|
||||
|
||||
|
||||
def run_bodies(path):
|
||||
"""Every `run:` block's text, with its line number."""
|
||||
with open(path, encoding="utf-8") as fh:
|
||||
lines = fh.readlines()
|
||||
|
||||
out = []
|
||||
i = 0
|
||||
while i < len(lines):
|
||||
m = re.match(r"^(\s*)run:\s*\|", lines[i])
|
||||
if not m:
|
||||
i += 1
|
||||
continue
|
||||
indent = len(m.group(1))
|
||||
start = i + 1
|
||||
body = []
|
||||
i += 1
|
||||
while i < len(lines):
|
||||
line = lines[i]
|
||||
if line.strip() and (len(line) - len(line.lstrip())) <= indent:
|
||||
break
|
||||
body.append(line)
|
||||
i += 1
|
||||
out.append((start + 1, "".join(body)))
|
||||
return out
|
||||
|
||||
|
||||
class TestNoExpressionInterpolationIntoShell:
|
||||
"""`${{ ... }}` inside a `run:` body is spliced into the SCRIPT TEXT.
|
||||
|
||||
This is not theoretical. `echo "${{ steps.report.outputs.body }}"` in the compat
|
||||
workflow pasted the report -- which is full of backticks -- straight into bash,
|
||||
which promptly ran `create-snapshot`, `def` and `async` as commands. And because
|
||||
that report is built from iX's middleware source, anything landing in their tree
|
||||
would have executed on our runner.
|
||||
|
||||
The rule: files for data, `env:` for scalars. `env:` is safe because the runner
|
||||
sets the variable rather than pasting it into the script.
|
||||
"""
|
||||
|
||||
@pytest.mark.parametrize("path", workflow_files(), ids=os.path.basename)
|
||||
def test_no_github_expression_in_a_run_body(self, path):
|
||||
offenders = []
|
||||
for lineno, body in run_bodies(path):
|
||||
for m in re.finditer(r"\$\{\{[^}]*\}\}", body):
|
||||
offenders.append(f"{os.path.basename(path)}:~{lineno}: {m.group(0)}")
|
||||
assert not offenders, (
|
||||
"GitHub/Gitea expressions interpolate into the shell script text, so "
|
||||
"backticks and $() in the value EXECUTE. Pass data via a file, or a "
|
||||
"scalar via `env:`.\n " + "\n ".join(offenders)
|
||||
)
|
||||
|
||||
|
||||
class TestBothForges:
|
||||
"""Gitea is canonical; GitHub is a mirror. Both run these files."""
|
||||
|
||||
def test_release_publishes_on_each_forge_exactly_once(self):
|
||||
with open(os.path.join(WORKFLOWS, "release.yml"), encoding="utf-8") as fh:
|
||||
src = fh.read()
|
||||
# One step gated ON github.com, one gated OFF it. Without the pair, a release
|
||||
# either double-publishes or silently never publishes on the canonical host.
|
||||
assert "if: ${{ contains(github.server_url, 'github.com') }}" in src
|
||||
assert "if: ${{ !contains(github.server_url, 'github.com') }}" in src
|
||||
|
||||
def test_compat_files_an_issue_on_each_forge(self):
|
||||
with open(os.path.join(WORKFLOWS, "compat.yml"), encoding="utf-8") as fh:
|
||||
src = fh.read()
|
||||
assert "file a bug report (GitHub)" in src
|
||||
assert "file a bug report (Gitea)" in src
|
||||
|
||||
|
||||
class TestCompatCannotSilentlyPass:
|
||||
def test_the_exit_code_is_captured_not_swallowed(self):
|
||||
# Actions runs `bash -e`: `cmd > out` followed by `echo $?` never reaches the
|
||||
# echo, so the "a shipped release is broken" signal would be lost and the job
|
||||
# would go green while users were broken.
|
||||
with open(os.path.join(WORKFLOWS, "compat.yml"), encoding="utf-8") as fh:
|
||||
src = fh.read()
|
||||
assert "|| rc=$?" in src
|
||||
assert "shipped_broken=$rc" in src
|
||||
|
||||
def test_a_broken_shipped_release_fails_the_job(self):
|
||||
with open(os.path.join(WORKFLOWS, "compat.yml"), encoding="utf-8") as fh:
|
||||
src = fh.read()
|
||||
assert "steps.check.outputs.shipped_broken != '0'" in src
|
||||
+993
@@ -0,0 +1,993 @@
|
||||
#!/usr/bin/env python3
|
||||
"""What this patch assumes about middlewared -- written down, and checkable.
|
||||
|
||||
WHY THIS EXISTS
|
||||
---------------
|
||||
This patch appends code to middlewared's own modules. middlewared has no stability
|
||||
contract: it is internal API, and iX may reshape it in any release. When they do,
|
||||
the patch does not politely decline -- it breaks a backup, possibly silently, which
|
||||
is the worst thing a backup tool can do.
|
||||
|
||||
It has already happened. TrueNAS 26 rewrites the whole cloud_backup path from
|
||||
async to synchronous:
|
||||
|
||||
25.10: async def create_snapshot(...) / await create_snapshot(...)
|
||||
26.0: def create_snapshot(...) / create_snapshot(...)
|
||||
|
||||
Every block the nested module injects is an `async def` wrapping an `await`ed
|
||||
original. On 26 that unpacks a coroutine object instead of a tuple. Nobody would
|
||||
have found out until a restore failed.
|
||||
|
||||
So the assumptions are written down here, once, and checked in two places:
|
||||
|
||||
* .github/workflows/compat.yml runs `--ref` against TrueNAS's *unreleased*
|
||||
branches (master, the newest BETA/RC) on a schedule, and opens a bug report
|
||||
the day iX breaks us -- while it is still a beta, not after it ships.
|
||||
|
||||
* patch/apply.sh runs `--tree` against the middlewared *actually installed*, at
|
||||
every boot, and REFUSES to patch a module whose assumptions no longer hold.
|
||||
That is the guarantee: an unpatched module means stock TrueNAS (Storj only,
|
||||
but working). A patched-anyway module means broken backups. Declining is
|
||||
always the better failure.
|
||||
|
||||
The two modules are checked independently, because they fail independently: on 26
|
||||
the providers module (B2/S3) only touches synchronous symbols and survives, while
|
||||
the nested module does not.
|
||||
|
||||
python3 tools/compat.py --tree /usr/lib/python3/dist-packages/middlewared
|
||||
python3 tools/compat.py --ref release/26.0.0-BETA.3
|
||||
python3 tools/compat.py --ref master --json
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import ast
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import urllib.request
|
||||
|
||||
PROVIDERS = "providers"
|
||||
NESTED = "nested"
|
||||
|
||||
RAW = "https://raw.githubusercontent.com/truenas/middleware/{ref}/src/middlewared/middlewared/{path}"
|
||||
|
||||
_TIMEOUT = 30
|
||||
|
||||
|
||||
class Assumption:
|
||||
"""One thing that must be true of middlewared, or a module cannot be applied.
|
||||
|
||||
`is_async=None` means "do not care". Everywhere else it is stated explicitly,
|
||||
because asyncness is exactly the axis TrueNAS 26 changed and a checker that
|
||||
ignored it would have passed a build that breaks every backup.
|
||||
"""
|
||||
|
||||
def __init__(self, ident, module, path, symbol, *, kind="function",
|
||||
is_async=None, params=None, forwards=False, why=""):
|
||||
self.id = ident
|
||||
self.module = module
|
||||
self.path = path
|
||||
self.symbol = symbol
|
||||
self.kind = kind
|
||||
self.is_async = is_async
|
||||
#: The positional parameters the patch passes, in order.
|
||||
self.params = params or []
|
||||
#: True if the wrapper takes *args/**kwargs and forwards the rest. Then a
|
||||
#: trailing parameter that iX adds or removes is harmless, and only the
|
||||
#: leading `params` must still match.
|
||||
self.forwards = forwards
|
||||
self.why = why
|
||||
|
||||
|
||||
#: Everything patch/apply.sh's injected blocks depend on. Derived from the blocks
|
||||
#: themselves -- if you add a block, add its assumptions here or the checker is
|
||||
#: decoration.
|
||||
ASSUMPTIONS = [
|
||||
# ── providers (B2/S3). Touches only synchronous symbols. ──────────────────
|
||||
Assumption(
|
||||
"b2-remote-class", PROVIDERS, "rclone/remote/b2.py", "B2RcloneRemote",
|
||||
kind="class",
|
||||
why="B2_BLOCK sets .get_restic_config and .restic on this class",
|
||||
),
|
||||
Assumption(
|
||||
"restic-config-fn", PROVIDERS, "plugins/cloud_backup/restic.py",
|
||||
"get_restic_config", is_async=False, params=["cloud_backup"],
|
||||
why="RESTIC_BLOCK wraps it to rewrite the repo URL; it calls the original "
|
||||
"WITHOUT await, so it must stay synchronous",
|
||||
),
|
||||
Assumption(
|
||||
"restic-config-class", PROVIDERS, "plugins/cloud_backup/restic.py",
|
||||
"ResticConfig", kind="class",
|
||||
why="RESTIC_BLOCK does dataclasses.replace(result, cmd=...) on what "
|
||||
"get_restic_config returns",
|
||||
),
|
||||
|
||||
# ── nested snapshots. Every block here is an async wrapper. ───────────────
|
||||
# is_async is deliberately NOT asserted on these three. The patch now injects an
|
||||
# async OR a sync wrapper to match whichever the installed middleware declares
|
||||
# (TrueNAS <= 25.10 is async; 26 rewrote them synchronous), so asyncness is a
|
||||
# thing to DETECT, not a thing to require -- see async_flavour(). What must still
|
||||
# hold is the shape: same name, same leading positional parameters.
|
||||
Assumption(
|
||||
"create-snapshot", NESTED, "plugins/cloud/snapshot.py", "create_snapshot",
|
||||
params=["middleware", "path", "name"],
|
||||
why="SNAPSHOT_BLOCK wraps it and returns (snapshot, staging_root) instead of "
|
||||
"(snapshot, snap_path)",
|
||||
),
|
||||
Assumption(
|
||||
"crud-mixin-validate", NESTED, "plugins/cloud/crud.py",
|
||||
"CloudTaskServiceMixin._validate",
|
||||
kind="method", params=["self", "app", "verrors", "name", "data"],
|
||||
why="CRUD_BLOCK wraps it to drop the no-further-nesting error",
|
||||
),
|
||||
Assumption(
|
||||
# SYNC_BLOCK's wrapper is (middleware, job, cloud_backup, *args, **kwargs) and
|
||||
# forwards the rest, precisely because iX keeps changing the tail: 24.10 and
|
||||
# 25.04 have `(…, dry_run)`, 25.10 added `rate_limit`. Only the leading three
|
||||
# are named by the patch, so only they have to hold.
|
||||
"restic-backup", NESTED, "plugins/cloud_backup/sync.py", "restic_backup",
|
||||
forwards=True,
|
||||
params=["middleware", "job", "cloud_backup"],
|
||||
why="SYNC_BLOCK wraps it to tear down bind mounts in a finally",
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
class MiddlewareCall:
|
||||
"""A middlewared METHOD the injected code calls at runtime.
|
||||
|
||||
THIS CLASS OF ASSUMPTION IS WHY THE CHECKER EXISTS, AND IT WAS THE ONE MISSING.
|
||||
|
||||
The manifest above records the symbols the patch *wraps*. It said nothing about
|
||||
the methods the patch *calls* -- and that gap hid two separate TrueNAS 26 breaks
|
||||
that both pass every other check:
|
||||
|
||||
* `get_dataset_recursive()` was deleted from plugins/cloud/snapshot.py, and the
|
||||
injected block called it out of the host module's namespace (now vendored).
|
||||
* plugins/zfs_/dataset.py and plugins/zfs_/snapshot.py were DELETED outright,
|
||||
taking `zfs.dataset.query`, `zfs.snapshot.query` and `zfs.snapshot.delete`
|
||||
with them. 26 uses filesystem.statfs and zfs.resource.* instead.
|
||||
|
||||
Nothing about the five cloud_backup files reveals that. The patch would apply
|
||||
perfectly, and then the FIRST BACKUP would fail -- or, far worse, succeed at
|
||||
snapshotting and fail at `zfs.snapshot.delete`, orphaning one snapshot per
|
||||
descendant dataset (250 on a real pool) on every single run, forever.
|
||||
|
||||
A method is present when some plugin file declares its namespace AND defines it.
|
||||
If iX merely MOVES a method to a different file we report BROKEN wrongly, and the
|
||||
module declines to apply -- costing a feature, not a backup. That asymmetry is
|
||||
the whole design: declining is always the cheaper mistake.
|
||||
"""
|
||||
|
||||
def __init__(self, ident, module, method, path, why=""):
|
||||
self.id = ident
|
||||
self.module = module
|
||||
self.method = method # "zfs.snapshot.delete"
|
||||
self.path = path # plugin file that declares it
|
||||
self.why = why
|
||||
|
||||
@property
|
||||
def namespace(self):
|
||||
return self.method.rsplit(".", 1)[0]
|
||||
|
||||
@property
|
||||
def name(self):
|
||||
return self.method.rsplit(".", 1)[1]
|
||||
|
||||
|
||||
#: Every middlewared method the nested module calls at runtime.
|
||||
MIDDLEWARE_CALLS = [
|
||||
MiddlewareCall(
|
||||
"call-zfs-dataset-query", NESTED, "zfs.dataset.query",
|
||||
"plugins/zfs_/dataset.py",
|
||||
why="SNAPSHOT_BLOCK enumerates FILESYSTEM datasets to build the staging plan",
|
||||
),
|
||||
MiddlewareCall(
|
||||
"call-zfs-snapshot-delete", NESTED, "zfs.snapshot.delete",
|
||||
"plugins/zfs_/snapshot.py",
|
||||
why="delete_snapshot_tree() sweeps the recursive snapshot. Without it every "
|
||||
"run orphans one snapshot per descendant dataset (250 on a real pool)",
|
||||
),
|
||||
MiddlewareCall(
|
||||
"call-zfs-snapshot-query", NESTED, "zfs.snapshot.query",
|
||||
"plugins/zfs_/snapshot.py",
|
||||
why="delete_snapshot_tree()'s fallback sweep enumerates the tree by name",
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
def check_call(c: MiddlewareCall, src: str | None) -> tuple[str, str | None]:
|
||||
"""Is `c.method` still registered by middlewared?"""
|
||||
if src is None:
|
||||
return "broken", (
|
||||
f"{c.path} no longer exists, so `{c.method}` is gone"
|
||||
)
|
||||
|
||||
try:
|
||||
tree = ast.parse(_stock(src))
|
||||
except SyntaxError as e:
|
||||
return "unknown", f"{c.path} does not parse: {e}"
|
||||
|
||||
# namespace = 'zfs.snapshot' on some Service class in this file...
|
||||
namespaces = {
|
||||
n.value.value
|
||||
for n in ast.walk(tree)
|
||||
if isinstance(n, ast.Assign)
|
||||
and isinstance(n.value, ast.Constant)
|
||||
and isinstance(n.value.value, str)
|
||||
and any(isinstance(t, ast.Name) and t.id == "namespace" for t in n.targets)
|
||||
}
|
||||
if c.namespace not in namespaces:
|
||||
return "broken", (
|
||||
f"{c.path} no longer declares namespace {c.namespace!r} "
|
||||
f"(found: {sorted(namespaces) or 'none'}), so `{c.method}` is gone"
|
||||
)
|
||||
|
||||
# ...and it defines the method.
|
||||
#
|
||||
# A CRUDService exposes `create`/`update`/`delete` from methods NAMED
|
||||
# `do_create`/`do_update`/`do_delete`. Both spellings are live right now:
|
||||
# 24.10 and 25.04 declare `do_delete`, 25.10 renamed it to `delete`, and all
|
||||
# three answer to `zfs.snapshot.delete`. Accepting only the literal name reported
|
||||
# the two older releases as broken -- a false BROKEN that would have switched off
|
||||
# nested snapshots on boxes where they work.
|
||||
defined = {
|
||||
n.name for n in ast.walk(tree)
|
||||
if isinstance(n, ast.FunctionDef | ast.AsyncFunctionDef)
|
||||
}
|
||||
if c.name not in defined and f"do_{c.name}" not in defined:
|
||||
return "broken", f"{c.path} no longer defines `{c.method}`"
|
||||
|
||||
return "ok", None
|
||||
|
||||
|
||||
#: Things that mean iX has done the job themselves and the module should RETIRE,
|
||||
#: not break. Absence of the nesting guard = nested snapshots went native.
|
||||
#: `restic = True` already on B2RcloneRemote = B2 restic support went native.
|
||||
NATIVE_PROBES = {
|
||||
NESTED: (
|
||||
"plugins/cloud/crud.py",
|
||||
"no further nesting",
|
||||
False, # native when the phrase is ABSENT
|
||||
),
|
||||
PROVIDERS: (
|
||||
"rclone/remote/b2.py",
|
||||
"restic = True",
|
||||
True, # native when the phrase is PRESENT
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def _squash(text: str) -> str:
|
||||
"""Drop whitespace and quotes, so a phrase split across string literals matches.
|
||||
|
||||
Stock middleware writes the guard as an implicitly-concatenated literal:
|
||||
|
||||
verrors.add(f"{name}.snapshot", "This option is only available for "
|
||||
"datasets that have no further nesting")
|
||||
|
||||
A naive `"no further nesting" in source` is therefore FALSE on a version that
|
||||
very much has the guard -- and this probe's False means "TrueNAS supports it
|
||||
natively, retire the module". That is a silent, catastrophic misread: it would
|
||||
disable nested snapshots on every box that currently depends on them.
|
||||
|
||||
apply.sh already learned this the hard way and normalises the same way. Both
|
||||
now call this one function, which is the only reason they cannot drift apart
|
||||
again.
|
||||
"""
|
||||
return text.translate(str.maketrans("", "", " \t\n\r\"'"))
|
||||
|
||||
|
||||
# ── AST lookups ──────────────────────────────────────────────────────────────
|
||||
|
||||
_DEFS = (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef)
|
||||
|
||||
|
||||
def _defs_in(body):
|
||||
"""Definitions in `body`, descending into if/try/else/with.
|
||||
|
||||
A module-level `def` is not always at module level:
|
||||
|
||||
try:
|
||||
from .fast import create_snapshot
|
||||
except ImportError:
|
||||
async def create_snapshot(...): ...
|
||||
|
||||
Scanning only `tree.body` would say "no longer defines create_snapshot" -- a
|
||||
false BROKEN. And a false BROKEN is not a harmless over-caution here: it makes a
|
||||
module decline to apply on a box where it works perfectly.
|
||||
"""
|
||||
for node in body:
|
||||
if isinstance(node, _DEFS):
|
||||
yield node
|
||||
elif isinstance(node, ast.If | ast.Try | ast.With | ast.AsyncWith):
|
||||
yield from _defs_in(node.body)
|
||||
yield from _defs_in(getattr(node, "orelse", []))
|
||||
yield from _defs_in(getattr(node, "finalbody", []))
|
||||
for h in getattr(node, "handlers", []):
|
||||
yield from _defs_in(h.body)
|
||||
|
||||
|
||||
def _imports(tree, name):
|
||||
"""True if `name` is bound by an import -- i.e. re-exported from elsewhere."""
|
||||
for node in ast.walk(tree):
|
||||
if isinstance(node, ast.Import | ast.ImportFrom):
|
||||
for alias in node.names:
|
||||
if (alias.asname or alias.name.split(".")[0]) == name:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _find(tree, symbol):
|
||||
"""The def node for `name` or `Class.method`, or None."""
|
||||
if "." in symbol:
|
||||
cls_name, meth = symbol.split(".", 1)
|
||||
for node in _defs_in(tree.body):
|
||||
if isinstance(node, ast.ClassDef) and node.name == cls_name:
|
||||
for sub in _defs_in(node.body):
|
||||
if isinstance(sub, ast.FunctionDef | ast.AsyncFunctionDef) \
|
||||
and sub.name == meth:
|
||||
return sub
|
||||
return None
|
||||
|
||||
for node in _defs_in(tree.body):
|
||||
if node.name == symbol:
|
||||
return node
|
||||
return None
|
||||
|
||||
|
||||
def _positional(node):
|
||||
a = node.args
|
||||
return [p.arg for p in (*a.posonlyargs, *a.args)]
|
||||
|
||||
|
||||
def _signature_problem(node, symbol, want, forwards=False):
|
||||
"""Why `symbol`'s signature no longer supports how the patch calls it.
|
||||
|
||||
The injected blocks call the original POSITIONALLY and with a fixed arg list:
|
||||
|
||||
await _tc_orig_create_snapshot(middleware, path, name)
|
||||
_tc_orig_get_restic_config(cloud_backup)
|
||||
|
||||
So a name-subset test ("are these names still in there somewhere?") is not
|
||||
enough, and that is what this used to be. It passed a reorder, a keyword-only
|
||||
conversion, and an added required parameter -- each of which is a TypeError or,
|
||||
worse, silently correct-looking with the arguments swapped.
|
||||
|
||||
The realistic one is not hypothetical: on `master`, iX already renamed
|
||||
get_restic_config's parameter and added a second. That function is rebound
|
||||
module-wide by RESTIC_BLOCK, so a wrong wrapper there kills EVERY TrueCloud
|
||||
task -- Storj included, for users who never wanted this patch's features.
|
||||
"""
|
||||
have = _positional(node)
|
||||
n = len(want)
|
||||
|
||||
if have[:n] != want:
|
||||
return (
|
||||
f"{symbol}{tuple(have)} — positional parameters changed; the patch "
|
||||
f"calls it as ({', '.join(want)})"
|
||||
)
|
||||
|
||||
# Extra parameters are fine only if they are optional -- the patch will not pass
|
||||
# them -- OR if the wrapper forwards *args/**kwargs, in which case whatever the
|
||||
# caller supplied is handed straight through. A new REQUIRED one that we neither
|
||||
# pass nor forward is a TypeError at the first backup.
|
||||
args = node.args
|
||||
required = len(have) - len(args.defaults)
|
||||
if required > n and not forwards:
|
||||
return (
|
||||
f"{symbol} now requires {', '.join(have[n:required])} — the patch does "
|
||||
f"not pass it"
|
||||
)
|
||||
|
||||
req_kwonly = [
|
||||
k.arg for k, d in zip(args.kwonlyargs, args.kw_defaults, strict=False)
|
||||
if d is None
|
||||
]
|
||||
if req_kwonly and not forwards:
|
||||
return (
|
||||
f"{symbol} now requires keyword-only {', '.join(req_kwonly)} — the "
|
||||
f"patch does not pass it"
|
||||
)
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def check_source(a: Assumption, src: str | None) -> tuple[str, str | None]:
|
||||
"""("ok"|"broken"|"unknown", detail).
|
||||
|
||||
"unknown" exists so that "I cannot inspect this" is never reported as "this is
|
||||
broken". Only "broken" makes a module decline to apply, and declining wrongly
|
||||
breaks a box that was working.
|
||||
"""
|
||||
if src is None:
|
||||
return "broken", f"{a.path} does not exist"
|
||||
|
||||
try:
|
||||
tree = ast.parse(src)
|
||||
except SyntaxError as e:
|
||||
return "unknown", f"{a.path} does not parse: {e}"
|
||||
|
||||
node = _find(tree, a.symbol)
|
||||
if node is None:
|
||||
root = a.symbol.split(".", 1)[0]
|
||||
if _imports(tree, root):
|
||||
# Re-exported: `from ._impl import get_restic_config`. The name is still
|
||||
# there and the patch's rebinding still works; we simply cannot see the
|
||||
# signature from here. Refusing to apply over a refactor that changed
|
||||
# nothing would be worse than not checking.
|
||||
return "unknown", (
|
||||
f"{a.path} re-exports {root} from another module; "
|
||||
f"cannot verify its signature here"
|
||||
)
|
||||
return "broken", f"{a.path} no longer defines {a.symbol}"
|
||||
|
||||
if a.kind == "class":
|
||||
if not isinstance(node, ast.ClassDef):
|
||||
return "broken", f"{a.symbol} is no longer a class"
|
||||
return "ok", None
|
||||
|
||||
if isinstance(node, ast.ClassDef):
|
||||
return "broken", f"{a.symbol} is a class, expected a function"
|
||||
|
||||
got_async = isinstance(node, ast.AsyncFunctionDef)
|
||||
if a.is_async is not None and got_async != a.is_async:
|
||||
want = "async def" if a.is_async else "def"
|
||||
got = "async def" if got_async else "def"
|
||||
return "broken", (
|
||||
f"{a.symbol} is now `{got}`, the patch requires `{want}` ({a.path})"
|
||||
)
|
||||
|
||||
problem = _signature_problem(node, a.symbol, a.params, a.forwards)
|
||||
return ("broken", problem) if problem else ("ok", None)
|
||||
|
||||
|
||||
# ── sources ──────────────────────────────────────────────────────────────────
|
||||
|
||||
class Unreadable(Exception):
|
||||
"""The source could not be READ. That is not the same as it not existing.
|
||||
|
||||
Folding these together is how a network blip becomes "iX deleted six files",
|
||||
which becomes "both modules are broken", which becomes a bug report, a red
|
||||
support matrix pushed to the README, and -- on a real box -- a module declining
|
||||
to apply. A transient failure must never be able to say anything about
|
||||
middleware.
|
||||
"""
|
||||
|
||||
|
||||
def _fetch(ref: str, path: str) -> str | None:
|
||||
"""Source at `ref`, None if iX genuinely does not have that file (404).
|
||||
|
||||
Raises Unreadable for anything else: rate limits (the matrix makes ~30
|
||||
unauthenticated requests per run and 429 is a real outcome), DNS, timeouts.
|
||||
"""
|
||||
url = RAW.format(ref=ref, path=path)
|
||||
try:
|
||||
with urllib.request.urlopen(url, timeout=_TIMEOUT) as r: # noqa: S310
|
||||
if r.status == 404:
|
||||
return None
|
||||
if r.status != 200:
|
||||
raise Unreadable(f"{url} -> HTTP {r.status}")
|
||||
return r.read().decode("utf-8", "replace")
|
||||
except urllib.error.HTTPError as e:
|
||||
if e.code == 404:
|
||||
return None # the file really is gone
|
||||
raise Unreadable(f"{url} -> HTTP {e.code}") from e
|
||||
except Unreadable:
|
||||
raise
|
||||
except Exception as e:
|
||||
raise Unreadable(f"{url} -> {e!r}") from e
|
||||
|
||||
|
||||
def _read(root: str, path: str) -> str | None:
|
||||
full = os.path.join(root, *path.split("/"))
|
||||
try:
|
||||
with open(full, encoding="utf-8") as fh:
|
||||
return fh.read()
|
||||
except FileNotFoundError:
|
||||
return None # genuinely absent
|
||||
except OSError as e:
|
||||
raise Unreadable(f"{full} -> {e!r}") from e # permissions, I/O, ...
|
||||
|
||||
|
||||
def _stock(text: str) -> str:
|
||||
"""Only the part of the file that iX wrote.
|
||||
|
||||
Our own blocks are appended after the MARKER, and they quote the very strings
|
||||
the probes look for -- B2_BLOCK literally writes `B2RcloneRemote.restic = True`
|
||||
into b2.py, and CRUD_BLOCK quotes the "no further nesting" message it filters
|
||||
on. Scanning the whole file on an already-patched box therefore finds OUR text
|
||||
and concludes TrueNAS went native, i.e. "retire the module". apply.sh has always
|
||||
cut at the marker for exactly this reason; compat.py did not, so the one command
|
||||
its own docstring recommends for a live box (`--tree /usr/lib/.../middlewared`)
|
||||
reported providers as native on every patched machine.
|
||||
"""
|
||||
return text.split("\n# TRUECLOUD_PATCH", 1)[0]
|
||||
|
||||
|
||||
def check(loader, modules=None) -> dict:
|
||||
"""Check every assumption. `loader(path) -> source|None`, may raise Unreadable.
|
||||
|
||||
Returns {module: {"ok", "native", "unknown", "problems"}}.
|
||||
|
||||
`unknown` means the sources could not be READ -- a rate limit, a timeout, an
|
||||
unreadable tree. It is NOT `ok` and it is emphatically NOT `broken`: nothing may
|
||||
act on a verdict derived from a failed download.
|
||||
"""
|
||||
modules = modules or [PROVIDERS, NESTED]
|
||||
cache = {}
|
||||
|
||||
def src(path):
|
||||
if path not in cache:
|
||||
cache[path] = loader(path)
|
||||
return cache[path]
|
||||
|
||||
out = {
|
||||
m: {"ok": True, "native": False, "unknown": False, "problems": []}
|
||||
for m in modules
|
||||
}
|
||||
|
||||
for a in ASSUMPTIONS:
|
||||
if a.module not in out:
|
||||
continue
|
||||
try:
|
||||
text = src(a.path)
|
||||
except Unreadable as e:
|
||||
out[a.module]["unknown"] = True
|
||||
out[a.module]["problems"].append({
|
||||
"id": a.id, "detail": f"could not read {a.path}: {e}", "why": a.why,
|
||||
})
|
||||
continue
|
||||
|
||||
status, detail = check_source(a, text if text is None else _stock(text))
|
||||
if status == "broken":
|
||||
out[a.module]["ok"] = False
|
||||
out[a.module]["problems"].append({
|
||||
"id": a.id, "detail": detail, "why": a.why,
|
||||
})
|
||||
elif status == "unknown":
|
||||
out[a.module]["unknown"] = True
|
||||
out[a.module]["problems"].append({
|
||||
"id": a.id, "detail": detail, "why": a.why,
|
||||
})
|
||||
|
||||
# The methods the injected code CALLS, not just the symbols it wraps.
|
||||
for c in MIDDLEWARE_CALLS:
|
||||
if c.module not in out:
|
||||
continue
|
||||
try:
|
||||
text = src(c.path)
|
||||
except Unreadable as e:
|
||||
out[c.module]["unknown"] = True
|
||||
out[c.module]["problems"].append({
|
||||
"id": c.id, "detail": f"could not read {c.path}: {e}", "why": c.why,
|
||||
})
|
||||
continue
|
||||
|
||||
status, detail = check_call(c, text)
|
||||
if status == "broken":
|
||||
out[c.module]["ok"] = False
|
||||
out[c.module]["problems"].append({
|
||||
"id": c.id, "detail": detail, "why": c.why,
|
||||
})
|
||||
elif status == "unknown":
|
||||
out[c.module]["unknown"] = True
|
||||
out[c.module]["problems"].append({
|
||||
"id": c.id, "detail": detail, "why": c.why,
|
||||
})
|
||||
|
||||
for module, (path, phrase, native_when_present) in NATIVE_PROBES.items():
|
||||
if module not in out:
|
||||
continue
|
||||
try:
|
||||
text = src(path)
|
||||
except Unreadable:
|
||||
out[module]["unknown"] = True
|
||||
continue
|
||||
if text is None:
|
||||
continue
|
||||
present = _squash(phrase) in _squash(_stock(text))
|
||||
out[module]["native"] = (present == native_when_present)
|
||||
|
||||
# `ok` is cleared ONLY by a definite violation, so "unknown" never needs to
|
||||
# repair it -- and must not: a module with one unreadable file AND one proven
|
||||
# broken assumption is broken, not unknown.
|
||||
return out
|
||||
|
||||
|
||||
#: The three stock symbols the nested module wraps. TrueNAS <= 25.10 declares them
|
||||
#: `async def`; TrueNAS 26 rewrote them synchronous. apply.sh injects the wrapper
|
||||
#: that matches, so this is the question it has to answer at every boot.
|
||||
NESTED_WRAPPED = [
|
||||
("plugins/cloud/snapshot.py", "create_snapshot"),
|
||||
("plugins/cloud/crud.py", "CloudTaskServiceMixin._validate"),
|
||||
("plugins/cloud_backup/sync.py", "restic_backup"),
|
||||
]
|
||||
|
||||
|
||||
def async_flavour(loader) -> bool | None:
|
||||
"""Is the installed cloud_backup path async? True, False, or None if unclear.
|
||||
|
||||
None means "do not patch": either a symbol is missing, or -- the case worth
|
||||
naming -- the three DISAGREE. A middleware caught half-converted is one this
|
||||
patch has never seen, and guessing a flavour there means injecting an `async def`
|
||||
that a synchronous caller unpacks as a tuple. Declining costs a feature; guessing
|
||||
costs a backup.
|
||||
"""
|
||||
flavours = set()
|
||||
for path, symbol in NESTED_WRAPPED:
|
||||
try:
|
||||
src = loader(path)
|
||||
except Unreadable:
|
||||
return None
|
||||
if src is None:
|
||||
return None
|
||||
try:
|
||||
tree = ast.parse(_stock(src))
|
||||
except SyntaxError:
|
||||
return None
|
||||
|
||||
node = _find(tree, symbol)
|
||||
if not isinstance(node, ast.FunctionDef | ast.AsyncFunctionDef):
|
||||
return None
|
||||
flavours.add(isinstance(node, ast.AsyncFunctionDef))
|
||||
|
||||
return flavours.pop() if len(flavours) == 1 else None
|
||||
|
||||
|
||||
def async_flavour_tree(root: str) -> bool | None:
|
||||
return async_flavour(lambda p: _read(root, p))
|
||||
|
||||
|
||||
def check_ref(ref: str, modules=None) -> dict:
|
||||
return check(lambda p: _fetch(ref, p), modules)
|
||||
|
||||
|
||||
def check_tree(root: str, modules=None) -> dict:
|
||||
return check(lambda p: _read(root, p), modules)
|
||||
|
||||
|
||||
# ── which TrueNAS versions to check ──────────────────────────────────────────
|
||||
|
||||
REPO = "https://github.com/truenas/middleware"
|
||||
|
||||
#: TrueCloud Backup -- the restic-based cloud_backup this patch extends -- was
|
||||
#: introduced in 24.10. In 24.04 the modules simply do not exist (404), which the
|
||||
#: checker would otherwise report as three separate "broken assumptions" for a
|
||||
#: feature that was never there.
|
||||
OLDEST = (24, 10)
|
||||
|
||||
#: BETA < RC < shipped. Without this, "26.0.0-BETA.1" and "26.0.0-BETA.3" both
|
||||
#: reduce to (26,0,0) and the matrix silently reports whichever was seen first --
|
||||
#: which is how it first showed BETA.1 while BETA.3 was the one to worry about.
|
||||
_STAGE = {"BETA": 0, "RC": 1}
|
||||
_SHIPPED = 2
|
||||
|
||||
|
||||
def _version_of(name: str):
|
||||
"""Sortable version of 'release/26.0.0-BETA.3' or 'TS-25.10.4'. None if junk.
|
||||
|
||||
Returns ((major, minor, ...), stage_rank, stage_number).
|
||||
"""
|
||||
tail = name.split("/", 1)[1] if "/" in name else name
|
||||
tail = tail.removeprefix("TS-")
|
||||
|
||||
core, _, suffix = tail.partition("-")
|
||||
try:
|
||||
version = tuple(int(p) for p in core.split("."))
|
||||
except ValueError:
|
||||
return None
|
||||
if len(version) < 2:
|
||||
return None
|
||||
|
||||
if not suffix:
|
||||
return version, _SHIPPED, 0
|
||||
|
||||
stage, _, num = suffix.partition(".")
|
||||
rank = _STAGE.get(stage.upper())
|
||||
if rank is None:
|
||||
return None # not a release line we understand
|
||||
return version, rank, int(num) if num.isdigit() else 0
|
||||
|
||||
|
||||
def _newest_per_line(names):
|
||||
"""Newest name on each (major, minor) line."""
|
||||
best = {}
|
||||
for name in names:
|
||||
v = _version_of(name)
|
||||
if not v or v[0][:2] < OLDEST:
|
||||
continue
|
||||
key = v[0][:2]
|
||||
if key not in best or v > best[key][0]:
|
||||
best[key] = (v, name)
|
||||
return [n for _, n in sorted(best.values())]
|
||||
|
||||
|
||||
def _ls_remote(remote, what):
|
||||
import subprocess
|
||||
|
||||
out = subprocess.run(
|
||||
["git", "ls-remote", what, "--refs", remote],
|
||||
capture_output=True, text=True, check=True, timeout=60,
|
||||
).stdout
|
||||
prefix = "refs/tags/" if what == "--tags" else "refs/heads/"
|
||||
return [
|
||||
line.split(prefix, 1)[1].strip()
|
||||
for line in out.splitlines() if prefix in line
|
||||
]
|
||||
|
||||
|
||||
def discover_refs(remote: str = REPO) -> list[str]:
|
||||
"""What to check: every shipped TrueNAS line, everything unreleased, and master.
|
||||
|
||||
Two sources, because they are authoritative for different things:
|
||||
|
||||
* SHIPPED comes from the `TS-*` TAGS. Those are what iX actually released.
|
||||
The `release/*` branches include mistakes -- `release/25.20.2.2` exists and
|
||||
25.20 is not a TrueNAS version -- and a typo branch in the matrix reads as
|
||||
a real supported release that we are silently broken on.
|
||||
|
||||
* UNRELEASED comes from the BRANCHES, because that is where a beta appears
|
||||
first: `release/26.0.0-BETA.3` had no tag yet while it was the newest beta.
|
||||
Catching breakage here, before it ships, is the whole point of this file.
|
||||
"""
|
||||
tags = _ls_remote(remote, "--tags")
|
||||
heads = _ls_remote(remote, "--heads")
|
||||
|
||||
shipped = _newest_per_line([
|
||||
t for t in tags if t.startswith("TS-") and "-BETA" not in t and "-RC" not in t
|
||||
])
|
||||
|
||||
# A prerelease of a line that has ALREADY shipped is history, not a warning:
|
||||
# release/24.10-RC.2 still exists, and the nested module does not apply to it,
|
||||
# but 24.10 shipped long ago and TS-24.10.2.4 is fine. Reporting it would be a
|
||||
# standing red row in the matrix for a version nobody can install.
|
||||
shipped_lines = {_version_of(t)[0][:2] for t in shipped}
|
||||
upcoming = [
|
||||
h for h in _newest_per_line([
|
||||
h for h in heads
|
||||
if h.startswith("release/") and ("-BETA" in h or "-RC" in h)
|
||||
])
|
||||
if _version_of(h)[0][:2] not in shipped_lines
|
||||
]
|
||||
|
||||
return [*shipped, *upcoming, "master"]
|
||||
|
||||
|
||||
def is_unreleased(ref: str) -> bool:
|
||||
"""master and any BETA/RC. Breakage here is early warning, not an outage."""
|
||||
return ref == "master" or "-BETA" in ref or "-RC" in ref
|
||||
|
||||
|
||||
def matrix(refs=None, remote: str = REPO) -> list[dict]:
|
||||
"""Check every release line. Returns one row per ref."""
|
||||
rows = []
|
||||
for ref in (refs or discover_refs(remote)):
|
||||
result = check(lambda p, r=ref: _fetch(r, p))
|
||||
rows.append({
|
||||
"ref": ref,
|
||||
"unreleased": is_unreleased(ref),
|
||||
"modules": result,
|
||||
})
|
||||
return rows
|
||||
|
||||
|
||||
def _verdict(r: dict) -> str:
|
||||
"""BROKEN outranks native, which outranks unknown.
|
||||
|
||||
"native" used to win outright, which meant a module that was BOTH broken and
|
||||
apparently-native rendered as good news: green CI, no bug report, and a README
|
||||
row telling users the feature went native while it was in fact broken. A proven
|
||||
violation is the strongest signal here and must never be masked by a weaker one
|
||||
-- and the native probe is only a substring match on iX's source, so it is
|
||||
exactly the weaker one.
|
||||
"""
|
||||
if not r["ok"]:
|
||||
return "BROKEN"
|
||||
if r["native"]:
|
||||
return "native"
|
||||
if r["unknown"]:
|
||||
return "unknown"
|
||||
return "ok"
|
||||
|
||||
|
||||
def is_broken(r: dict) -> bool:
|
||||
return not r["ok"]
|
||||
|
||||
|
||||
#: Versions a human has actually run a backup on, with real data, on real hardware.
|
||||
#: This is NOT automatable and must never be inferred: everything else in this file
|
||||
#: is static analysis of iX's source, which proves the patch's assumptions hold --
|
||||
#: a strictly weaker claim than "a restore worked". Add a row only after doing it.
|
||||
HARDWARE_VERIFIED = {
|
||||
"25.10.4": "nested + providers; 252-snapshot recursive backup of /mnt/Tap, 18m",
|
||||
}
|
||||
|
||||
_LEGEND = """
|
||||
| verdict | meaning |
|
||||
| --- | --- |
|
||||
| **ok** | Every assumption the patch makes about middleware still holds. |
|
||||
| **BROKEN** | middleware changed underneath the patch. `apply.sh` **refuses to apply that module** on this version and leaves TrueNAS stock, so backups keep working — without the module's feature. |
|
||||
| **native** | TrueNAS does this itself now. The module retires; it is not a failure. |
|
||||
|
||||
"ok" means *the patch's assumptions hold*, checked automatically against iX's
|
||||
source. It does not mean a human ran a backup on it — that is the
|
||||
**Hardware-verified** column, which is filled in by hand and only by doing it.
|
||||
"""
|
||||
|
||||
|
||||
#: The README's matrix lives between these. CI regenerates it daily, so a table
|
||||
#: claiming the patch works on a TrueNAS that iX has since changed cannot survive
|
||||
#: for longer than a day -- a stale support matrix is not a stale doc, it is a lie
|
||||
#: to somebody deciding whether to trust this with their backups.
|
||||
ROOT_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
README = os.path.join(ROOT_DIR, "README.md")
|
||||
BEGIN = "<!-- BEGIN COMPAT MATRIX (generated by tools/compat.py --matrix --markdown) -->"
|
||||
END = "<!-- END COMPAT MATRIX -->"
|
||||
|
||||
|
||||
def update_readme(rows: list[dict], path: str = README) -> bool:
|
||||
"""Rewrite the README's matrix block. True if it changed.
|
||||
|
||||
Refuses if ANY row could not be fully checked. The published matrix is what a
|
||||
stranger reads before trusting this with their backups, and CI pushes it
|
||||
automatically -- so a rate limit or a DNS blip must never be able to repaint it.
|
||||
A stale-but-true table beats a fresh-but-invented one.
|
||||
"""
|
||||
unknown = [
|
||||
r["ref"] for r in rows
|
||||
if any(m["unknown"] for m in r["modules"].values())
|
||||
]
|
||||
if unknown:
|
||||
raise Unreadable(
|
||||
"not rewriting the matrix: could not fully check " + ", ".join(unknown)
|
||||
)
|
||||
|
||||
with open(path, encoding="utf-8") as fh:
|
||||
text = fh.read()
|
||||
|
||||
i, j = text.find(BEGIN), text.find(END)
|
||||
if i == -1 or j == -1:
|
||||
raise ValueError(f"{path} has no COMPAT MATRIX markers")
|
||||
|
||||
new = f"{BEGIN}\n{render_markdown(rows).rstrip()}\n{END}"
|
||||
old = text[i:j + len(END)]
|
||||
if old == new:
|
||||
return False
|
||||
|
||||
with open(path, "w", encoding="utf-8") as fh:
|
||||
fh.write(text[:i] + new + text[j + len(END):])
|
||||
return True
|
||||
|
||||
|
||||
def render_markdown(rows: list[dict]) -> str:
|
||||
"""The matrix, for the README."""
|
||||
out = [
|
||||
"| TrueNAS | B2/S3 providers | Nested snapshots | Hardware-verified |",
|
||||
"| --- | --- | --- | --- |",
|
||||
]
|
||||
for row in rows:
|
||||
m = row["modules"]
|
||||
ref = row["ref"]
|
||||
label = ref.removeprefix("TS-").removeprefix("release/")
|
||||
if row["unreleased"]:
|
||||
label = f"{label} _(unreleased)_"
|
||||
|
||||
cells = []
|
||||
for mod in (PROVIDERS, NESTED):
|
||||
v = _verdict(m[mod])
|
||||
cells.append({
|
||||
"ok": "ok",
|
||||
"BROKEN": "**BROKEN**",
|
||||
"native": "native",
|
||||
"unknown": "unknown",
|
||||
}[v])
|
||||
|
||||
version = ref.removeprefix("TS-")
|
||||
hw = HARDWARE_VERIFIED.get(version)
|
||||
out.append(f"| {label} | {cells[0]} | {cells[1]} | {hw or '—'} |")
|
||||
|
||||
return "\n".join(out) + "\n" + _LEGEND
|
||||
|
||||
|
||||
def render_matrix(rows: list[dict]) -> str:
|
||||
"""A support table.
|
||||
|
||||
Says "assumptions hold", not "works" -- this is static analysis of iX's source,
|
||||
which is a strictly weaker claim than having run a backup on the hardware. The
|
||||
hardware-verified column lives in COMPATIBILITY.md and is maintained by hand,
|
||||
because nothing else can honestly fill it in.
|
||||
"""
|
||||
w = max((len(r["ref"]) for r in rows), default=10)
|
||||
lines = [
|
||||
f"{'TrueNAS'.ljust(w)} {'providers':<10} {'nested':<10}",
|
||||
f"{'-' * w} {'-' * 10} {'-' * 10}",
|
||||
]
|
||||
for row in rows:
|
||||
m = row["modules"]
|
||||
lines.append(
|
||||
f"{row['ref'].ljust(w)} "
|
||||
f"{_verdict(m[PROVIDERS]):<10} {_verdict(m[NESTED]):<10}"
|
||||
)
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
# ── reporting ────────────────────────────────────────────────────────────────
|
||||
|
||||
def render(label: str, result: dict) -> str:
|
||||
lines = [f"TrueNAS middleware @ {label}", ""]
|
||||
for module, r in sorted(result.items()):
|
||||
if r["native"]:
|
||||
lines.append(
|
||||
f" [NATIVE] {module}: TrueNAS appears to support this natively "
|
||||
f"now — the module should be retired, not fixed."
|
||||
)
|
||||
elif r["ok"]:
|
||||
lines.append(f" [ok] {module}: all assumptions hold")
|
||||
else:
|
||||
lines.append(f" [BROKEN] {module}:")
|
||||
for p in r["problems"]:
|
||||
lines.append(f" - {p['detail']}")
|
||||
lines.append(f" why it matters: {p['why']}")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def main(argv):
|
||||
ap = argparse.ArgumentParser(description=__doc__.split("\n")[0])
|
||||
g = ap.add_mutually_exclusive_group(required=True)
|
||||
g.add_argument("--ref", help="a truenas/middleware git ref, e.g. master")
|
||||
g.add_argument("--tree", help="path to an installed middlewared package")
|
||||
g.add_argument("--matrix", action="store_true",
|
||||
help="check every TrueNAS release line, newest of each")
|
||||
ap.add_argument("--module", action="append", choices=[PROVIDERS, NESTED],
|
||||
help="check only this module (repeatable)")
|
||||
ap.add_argument("--json", action="store_true")
|
||||
ap.add_argument("--markdown", action="store_true",
|
||||
help="with --matrix: emit the table as markdown")
|
||||
ap.add_argument("--update-readme", action="store_true",
|
||||
help="with --matrix: rewrite the README's matrix block in place")
|
||||
args = ap.parse_args(argv[1:])
|
||||
|
||||
if args.matrix:
|
||||
rows = matrix()
|
||||
if args.json:
|
||||
print(json.dumps(rows, indent=2))
|
||||
elif args.markdown:
|
||||
print(render_markdown(rows))
|
||||
elif args.update_readme:
|
||||
changed = update_readme(rows)
|
||||
print("README.md updated" if changed else "README.md already current")
|
||||
else:
|
||||
print(render_matrix(rows))
|
||||
# A broken UNRELEASED line (master, -BETA, -RC) is a warning, not a build
|
||||
# failure -- it is exactly what we want to know early, and it is iX's tree
|
||||
# to change. compat.yml turns it into a bug report. A broken SHIPPED line
|
||||
# is a genuine failure: users are on it right now.
|
||||
shipped_broken = [
|
||||
r["ref"] for r in rows
|
||||
if not r["unreleased"]
|
||||
and any(is_broken(m) for m in r["modules"].values())
|
||||
]
|
||||
if shipped_broken:
|
||||
print(f"\nBROKEN on shipped releases: {', '.join(shipped_broken)}",
|
||||
file=sys.stderr)
|
||||
return 1
|
||||
return 0
|
||||
|
||||
label = args.ref or args.tree
|
||||
result = (check_ref(args.ref, args.module) if args.ref
|
||||
else check_tree(args.tree, args.module))
|
||||
|
||||
if args.json:
|
||||
print(json.dumps({"ref": label, "modules": result}, indent=2))
|
||||
else:
|
||||
print(render(label, result))
|
||||
|
||||
# Exit 1 if any module is broken. "Native" is not broken -- it is good news.
|
||||
return 1 if any(is_broken(r) for r in result.values()) else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv))
|
||||
@@ -0,0 +1,160 @@
|
||||
#!/usr/bin/env python3
|
||||
"""The barrier: a stable release must have been a release candidate first.
|
||||
|
||||
CHECKS
|
||||
------
|
||||
release_notes.py checks the *content* of the tree (versions agree, CHANGELOG has a
|
||||
non-empty section, nothing stranded under Unreleased). This module checks the
|
||||
*provenance* of the commit: was this exact code ever a release candidate, and did
|
||||
that candidate pass CI?
|
||||
|
||||
WHY
|
||||
---
|
||||
This repo cut twelve releases in a single day, several of them "fix the thing the
|
||||
last release broke". With an update alert live on every user's box, that is not
|
||||
iteration, it is nagging -- and it teaches people to ignore the alert that will one
|
||||
day carry a real security fix.
|
||||
|
||||
The rule that makes the bad path impossible:
|
||||
|
||||
A stable vX.Y.Z tag is only publishable if a vX.Y.Z-rcN tag points at the SAME
|
||||
commit, and that candidate's CI run passed.
|
||||
|
||||
Release candidates are invisible to users: update.sh and the alert source both take
|
||||
the newest plain vX.Y.Z tag, so an rc is never offered as an update. Debugging
|
||||
therefore happens across rc1, rc2, rc3 -- where it costs nobody anything -- instead
|
||||
of across v0.5.0, v0.5.1, v0.5.2, where it costs everybody an alert.
|
||||
|
||||
The commit must be *identical*, not merely an ancestor. "The rc passed, then I
|
||||
pushed one more little fix" is exactly the habit this exists to break.
|
||||
|
||||
python3 tools/release_gate.py v0.6.0 # exit 1 if not promotable
|
||||
python3 tools/release_gate.py v0.6.0 --next-rc # -> the rc tag to cut next
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
from release_notes import base_version, is_prerelease, normalise
|
||||
|
||||
|
||||
def _git(*args: str, cwd: str | None = None) -> str:
|
||||
return subprocess.run(
|
||||
["git", *args],
|
||||
cwd=cwd, capture_output=True, text=True, check=True,
|
||||
).stdout.strip()
|
||||
|
||||
|
||||
def rc_tags(version: str, cwd: str | None = None) -> list[str]:
|
||||
"""Every rc tag for this version, oldest first (rc2 sorts after rc1).
|
||||
|
||||
The glob is only a prefilter; the anchored regex decides. `v1.0.0-rc*` also
|
||||
matches `v1.0.0-rc1-hotfix`, which _rc_number reads as 0 -- so a tag the
|
||||
numbering logic does not understand could satisfy the barrier while never having
|
||||
been a release candidate.
|
||||
"""
|
||||
want = re.escape(normalise(base_version(version)))
|
||||
exact = re.compile(rf"^v{want}-rc\d+$")
|
||||
|
||||
out = _git("tag", "--list", f"v{normalise(base_version(version))}-rc*", cwd=cwd)
|
||||
tags = [t.strip() for t in out.splitlines() if exact.match(t.strip())]
|
||||
return sorted(tags, key=_rc_number)
|
||||
|
||||
|
||||
def _rc_number(tag: str) -> int:
|
||||
m = re.search(r"-rc(\d+)$", tag)
|
||||
return int(m.group(1)) if m else 0
|
||||
|
||||
|
||||
def next_rc(version: str, cwd: str | None = None) -> str:
|
||||
"""The next rc tag to cut: v0.6.0-rc1, then -rc2, ..."""
|
||||
existing = rc_tags(version, cwd=cwd)
|
||||
n = max((_rc_number(t) for t in existing), default=0) + 1
|
||||
return f"v{normalise(base_version(version))}-rc{n}"
|
||||
|
||||
|
||||
def commit_for(ref: str, cwd: str | None = None) -> str | None:
|
||||
try:
|
||||
return _git("rev-list", "-n", "1", ref, cwd=cwd)
|
||||
except subprocess.CalledProcessError:
|
||||
return None
|
||||
|
||||
|
||||
def check_promotable(version: str, cwd: str | None = None) -> list[str]:
|
||||
"""Every reason v<version> may not be cut as a stable release.
|
||||
|
||||
Empty list means the barrier is satisfied.
|
||||
|
||||
The commit under test is the tag's if it exists, and HEAD otherwise. Both are
|
||||
real: CI runs this AFTER the tag is pushed, and release.sh runs it BEFORE
|
||||
creating the tag -- which is the whole point, since refusing after the tag
|
||||
exists is too late to be a gate. Requiring the tag unconditionally made
|
||||
`release.sh --promote` impossible: it dies if the tag already exists, and the
|
||||
gate died if it did not, so the only way through was to hand-tag and bypass
|
||||
every check this file exists to enforce.
|
||||
"""
|
||||
if is_prerelease(version):
|
||||
return [] # candidates are what the barrier exists to encourage
|
||||
|
||||
want = normalise(version)
|
||||
tag = f"v{want}"
|
||||
|
||||
target = commit_for(tag, cwd=cwd)
|
||||
if target is None:
|
||||
target = commit_for("HEAD", cwd=cwd)
|
||||
if target is None:
|
||||
return ["cannot resolve a commit to release (no HEAD?)"]
|
||||
|
||||
candidates = rc_tags(want, cwd=cwd)
|
||||
if not candidates:
|
||||
return [
|
||||
f"{tag} was never a release candidate. Cut one first:\n"
|
||||
f" bash release.sh {want} --rc\n"
|
||||
f"Candidates are invisible to users -- debug there, not in a release."
|
||||
]
|
||||
|
||||
matching = [c for c in candidates if commit_for(c, cwd=cwd) == target]
|
||||
if not matching:
|
||||
newest = candidates[-1]
|
||||
return [
|
||||
f"{tag} points at {target[:12]}, but no release candidate does.\n"
|
||||
f" Candidates: {', '.join(candidates)}\n"
|
||||
f" Newest ({newest}) is at "
|
||||
f"{(commit_for(newest, cwd=cwd) or '?')[:12]}.\n"
|
||||
f"Code changed after the last candidate. That change is untested as a\n"
|
||||
f"release: cut {next_rc(want, cwd=cwd)} and promote THAT commit."
|
||||
]
|
||||
|
||||
return []
|
||||
|
||||
|
||||
def main(argv: list[str]) -> int:
|
||||
ap = argparse.ArgumentParser(description=__doc__.split("\n")[0])
|
||||
ap.add_argument("version", help="e.g. v0.6.0")
|
||||
ap.add_argument("--next-rc", action="store_true",
|
||||
help="print the next rc tag to cut, and exit")
|
||||
ap.add_argument("-C", dest="cwd", default=None, help="run git in this directory")
|
||||
args = ap.parse_args(argv[1:])
|
||||
|
||||
if args.next_rc:
|
||||
print(next_rc(args.version, cwd=args.cwd))
|
||||
return 0
|
||||
|
||||
problems = check_promotable(args.version, cwd=args.cwd)
|
||||
for p in problems:
|
||||
print(f"::error::{p}")
|
||||
if problems:
|
||||
return 1
|
||||
|
||||
print(f"v{normalise(args.version)} was a release candidate and may be promoted")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# Running `python3 tools/release_gate.py` already puts tools/ on sys.path[0],
|
||||
# which is what makes the `release_notes` import above resolve.
|
||||
sys.exit(main(sys.argv))
|
||||
+118
-3
@@ -37,10 +37,53 @@ _VERSION_RE = re.compile(r'^(?:VERSION=|__version__\s*=\s*)"([^"]+)"', re.M)
|
||||
_HEADING_RE = re.compile(r"^##\s+v?(\d+\.\d+\.\d+[^\s]*)", re.M)
|
||||
|
||||
|
||||
#: Work in progress lives here until a release promotes it. Batching through this
|
||||
#: section is what stops "tag, find bug, tag again" from becoming twelve releases.
|
||||
UNRELEASED = "Unreleased"
|
||||
|
||||
_UNRELEASED_RE = re.compile(r"^##\s+Unreleased\s*$", re.M | re.I)
|
||||
_RC_RE = re.compile(r"-(rc|beta|alpha)\d*$", re.I)
|
||||
|
||||
|
||||
def normalise(v: str) -> str:
|
||||
return v.strip().lstrip("v")
|
||||
|
||||
|
||||
def is_prerelease(tag: str) -> bool:
|
||||
"""True for v1.2.3-rc1 / -beta / -alpha. Those never reach users."""
|
||||
return bool(_RC_RE.search(tag.strip()))
|
||||
|
||||
|
||||
def base_version(tag: str) -> str:
|
||||
"""v1.2.3-rc2 -> 1.2.3"""
|
||||
return _RC_RE.sub("", normalise(tag))
|
||||
|
||||
|
||||
def unreleased_body(text: str) -> str:
|
||||
"""Content under `## Unreleased`, or "" if the section is absent/empty."""
|
||||
m = _UNRELEASED_RE.search(text)
|
||||
if not m:
|
||||
return ""
|
||||
rest = text[m.end():]
|
||||
nxt = _HEADING_RE.search(rest)
|
||||
return (rest[:nxt.start()] if nxt else rest).strip()
|
||||
|
||||
|
||||
def promote(text: str, version: str, date: str) -> str:
|
||||
"""Rename `## Unreleased` to `## vX.Y.Z — date`.
|
||||
|
||||
Refuses if the section is missing or empty: a release with nothing in it is a
|
||||
release nobody needed, and cutting one only trains people to ignore alerts.
|
||||
"""
|
||||
if not unreleased_body(text):
|
||||
raise ValueError(
|
||||
"CHANGELOG.md has no `## Unreleased` content — nothing to release. "
|
||||
"Add your changes there first."
|
||||
)
|
||||
m = _UNRELEASED_RE.search(text)
|
||||
return text[:m.start()] + f"## v{normalise(version)} — {date}" + text[m.end():]
|
||||
|
||||
|
||||
def script_versions(root: str = ROOT) -> dict[str, str]:
|
||||
"""VERSION= as declared by each script."""
|
||||
found = {}
|
||||
@@ -64,10 +107,15 @@ def changelog_versions(text: str) -> list[str]:
|
||||
def extract_notes(text: str, version: str) -> str:
|
||||
"""The body of one version's section, without its heading.
|
||||
|
||||
A release candidate resolves to its BASE version: v0.6.0-rc2 ships the same code
|
||||
as v0.6.0 and therefore the same notes, and the CHANGELOG only ever has the one
|
||||
section. Without this, the release workflow cut the tag, passed every gate, and
|
||||
then died extracting the body -- so the candidate existed but was never published.
|
||||
|
||||
Raises KeyError if the version has no section -- a release with an empty or
|
||||
wrong body is worse than a failed release.
|
||||
"""
|
||||
want = normalise(version)
|
||||
want = base_version(version)
|
||||
lines = text.splitlines()
|
||||
|
||||
start = None
|
||||
@@ -88,9 +136,67 @@ def extract_notes(text: str, version: str) -> str:
|
||||
return "\n".join(lines[start:end]).strip()
|
||||
|
||||
|
||||
# ── significance ──────────────────────────────────────────────────────────────
|
||||
# Used by the TrueNAS update alert to decide whether a release is worth bothering
|
||||
# anyone about. The CHANGELOG's own section headings are the signal: a release that
|
||||
# only has "### Docs" changed no code, and nobody should get an alert for a README.
|
||||
|
||||
_SECTION_RE = re.compile(r"^###\s+(.+?)\s*$", re.M)
|
||||
|
||||
#: Headings that mean "nothing about the running system changed".
|
||||
QUIET_SECTIONS = {"docs", "documentation"}
|
||||
|
||||
|
||||
def version_tuple(v: str) -> tuple:
|
||||
"""Sortable version. Pre-release suffixes are dropped, not ranked."""
|
||||
return tuple(int(x) for x in normalise(v).split("-")[0].split("."))
|
||||
|
||||
|
||||
def section_headings(body: str) -> list[str]:
|
||||
"""The `### ...` headings inside one version's body, lowercased."""
|
||||
return [h.strip().lower() for h in _SECTION_RE.findall(body)]
|
||||
|
||||
|
||||
def significance(text: str, current: str, latest: str):
|
||||
"""How much does upgrading `current` -> `latest` actually matter?
|
||||
|
||||
Returns ``(level, versions, headings)`` where level is one of:
|
||||
|
||||
"security" a release in the range has a Security section -> alert loudly
|
||||
"notable" something about the system changed -> alert quietly
|
||||
"docs" only documentation changed -> DO NOT alert
|
||||
|
||||
Considers every release in the range, not just the newest: a docs-only v0.4.2
|
||||
on top of a security-fixing v0.4.1 must still be reported as security.
|
||||
"""
|
||||
cur, lat = version_tuple(current), version_tuple(latest)
|
||||
|
||||
versions = [
|
||||
v for v in changelog_versions(text)
|
||||
if cur < version_tuple(v) <= lat
|
||||
]
|
||||
|
||||
headings = []
|
||||
for v in versions:
|
||||
try:
|
||||
headings.extend(section_headings(extract_notes(text, v)))
|
||||
except KeyError:
|
||||
continue
|
||||
|
||||
if any(h.startswith("security") for h in headings):
|
||||
return "security", versions, headings
|
||||
if [h for h in headings if h not in QUIET_SECTIONS]:
|
||||
return "notable", versions, headings
|
||||
return "docs", versions, headings
|
||||
|
||||
|
||||
def check(version: str, root: str = ROOT) -> list[str]:
|
||||
"""Every reason this version is not releasable. Empty list means it is."""
|
||||
want = normalise(version)
|
||||
"""Every reason this version is not releasable. Empty list means it is.
|
||||
|
||||
`version` may be a release candidate (v1.2.3-rc2); the scripts and CHANGELOG
|
||||
are checked against its BASE version, since an rc ships the same code.
|
||||
"""
|
||||
want = base_version(version)
|
||||
problems = []
|
||||
|
||||
versions = script_versions(root)
|
||||
@@ -116,6 +222,15 @@ def check(version: str, root: str = ROOT) -> list[str]:
|
||||
if not body:
|
||||
problems.append(f"CHANGELOG.md section for v{want} is empty")
|
||||
|
||||
# A stable release must not leave work stranded under `## Unreleased`. If it is
|
||||
# finished enough to ship, it belongs in the release; if it is not, the release
|
||||
# is premature. (An rc may legitimately have more work queued behind it.)
|
||||
if not is_prerelease(version) and unreleased_body(text):
|
||||
problems.append(
|
||||
"CHANGELOG.md still has content under `## Unreleased` — either include "
|
||||
"it in this release, or do not cut the release yet"
|
||||
)
|
||||
|
||||
return problems
|
||||
|
||||
|
||||
|
||||
+1
-1
@@ -3,7 +3,7 @@
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
VERSION="0.4.0"
|
||||
VERSION="0.6.1"
|
||||
|
||||
PATCH_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
_HOOK_COMMENT='TrueCloud provider patch (S3/B2)'
|
||||
|
||||
@@ -19,7 +19,7 @@
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
VERSION="0.4.0"
|
||||
VERSION="0.6.1"
|
||||
|
||||
PATCH_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
_PREV_FILE="$PATCH_DIR/.update_previous"
|
||||
@@ -46,9 +46,59 @@ Updating preserves your nested-snapshot opt-in setting either way.
|
||||
USAGE
|
||||
}
|
||||
|
||||
# An UNTRACKED file that the target tracks makes `git checkout` abort. The dirty-
|
||||
# tree check deliberately ignores untracked files, so this slips past it and the
|
||||
# checkout then dies mid-operation. Not hypothetical: a hand-copied
|
||||
# patch/wait_restart.sh blocked a pull on a real box exactly this way.
|
||||
#
|
||||
# Used by BOTH the update and the rollback path -- rolling back moves the tree too,
|
||||
# and would hit the identical failure.
|
||||
_abort_if_untracked_blockers() {
|
||||
local ref="$1" blocking
|
||||
|
||||
# Set intersection of {untracked, not ignored} and {tracked by the target}. Two
|
||||
# git calls, not one `ls-files --error-unmatch` per file in the target tree.
|
||||
# --exclude-standard is deliberate: git silently overwrites *ignored* files on
|
||||
# checkout, so those are not blockers — only untracked-and-not-ignored ones are.
|
||||
blocking="$(comm -12 \
|
||||
<(git ls-files --others --exclude-standard | sort) \
|
||||
<(git ls-tree -r --name-only "$ref" | sort) \
|
||||
| sed 's/^/ /')"
|
||||
|
||||
[ -n "$blocking" ] || return 0
|
||||
|
||||
echo "ERROR: these untracked files would be overwritten:" >&2
|
||||
printf '%s\n\n' "$blocking" >&2
|
||||
echo " They exist here but git does not track them — most likely hand-copied" >&2
|
||||
echo " or scp'd in. Move or delete them, then re-run." >&2
|
||||
|
||||
# "Delete update.sh, then re-run update.sh" is impossible. If the script itself
|
||||
# is a blocker, it was hand-copied in to bootstrap; the honest answer is to
|
||||
# bootstrap with git instead, which installs it properly.
|
||||
case "$blocking" in
|
||||
*update.sh*)
|
||||
echo "" >&2
|
||||
echo " update.sh itself is untracked here — you copied it in to bootstrap." >&2
|
||||
echo " Do that with git instead, once; it installs update.sh properly:" >&2
|
||||
echo "" >&2
|
||||
echo " rm -f $PATCH_DIR/update.sh" >&2
|
||||
echo " git -C $PATCH_DIR checkout $ref" >&2
|
||||
echo " bash $PATCH_DIR/install.sh" >&2
|
||||
echo "" >&2
|
||||
echo " Every later update is then just: bash update.sh" >&2
|
||||
;;
|
||||
esac
|
||||
exit 1
|
||||
}
|
||||
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--to) _target="${2:-}"; shift ;;
|
||||
--to)
|
||||
if [ -z "${2:-}" ]; then
|
||||
echo "ERROR: --to needs a tag, branch, or commit." >&2
|
||||
exit 1
|
||||
fi
|
||||
_target="$2"; shift ;;
|
||||
--main) _use_main=1 ;;
|
||||
--check) _check_only=1 ;;
|
||||
--rollback) _rollback=1 ;;
|
||||
@@ -86,9 +136,21 @@ fi
|
||||
|
||||
# A dirty tree means someone edited or scp'd files in place; merging over that
|
||||
# silently loses their changes, or conflicts halfway through.
|
||||
if [ -n "$(git status --porcelain --untracked-files=no)" ]; then
|
||||
#
|
||||
# CONTENT changes only. A mode-only change (100644 -> 100755) is not somebody's work
|
||||
# and must not block an update -- and it is not hypothetical: install.sh chmod +x's
|
||||
# these very scripts, so on any version where git recorded one as 100644, INSTALLING
|
||||
# dirtied the checkout and update.sh then refused to run. Install once, and updating
|
||||
# was blocked forever, with an error telling the user to `git checkout -- .` (which
|
||||
# merely undoes the exec bit so the next install can re-dirty it). A real box sat on
|
||||
# an old version for exactly this reason.
|
||||
#
|
||||
# `git diff --numstat` reports "0 0 file" for a mode-only change, so anything with a
|
||||
# nonzero insert or delete count is a genuine edit.
|
||||
_dirty=$(git diff --numstat HEAD -- . | awk '$1 != 0 || $2 != 0 { print $3 }')
|
||||
if [ -n "$_dirty" ]; then
|
||||
echo "ERROR: the working tree has uncommitted changes:" >&2
|
||||
git status --short --untracked-files=no >&2
|
||||
printf ' M %s\n' $_dirty >&2
|
||||
echo "" >&2
|
||||
echo " Refusing to update over them. Commit, stash, or discard them first:" >&2
|
||||
echo " git -C $PATCH_DIR checkout -- ." >&2
|
||||
@@ -103,8 +165,15 @@ if [ "$_rollback" -eq 1 ]; then
|
||||
exit 1
|
||||
fi
|
||||
_prev="$(cat "$_PREV_FILE")"
|
||||
if ! git rev-parse --verify --quiet "${_prev}^{commit}" >/dev/null; then
|
||||
echo "ERROR: recorded revision '$_prev' is not a valid commit." >&2
|
||||
echo " The history may have been rewritten. Pick a target explicitly:" >&2
|
||||
echo " bash update.sh --to <tag>" >&2
|
||||
exit 1
|
||||
fi
|
||||
_abort_if_untracked_blockers "$_prev"
|
||||
echo "Rolling back to $_prev ..."
|
||||
git checkout -q "$_prev"
|
||||
git checkout -q --detach "$_prev"
|
||||
echo "Reverted. Re-applying ..."
|
||||
echo ""
|
||||
bash "$PATCH_DIR/install.sh"
|
||||
@@ -128,7 +197,13 @@ else
|
||||
# correct while tags are created in ascending version order; it breaks the
|
||||
# moment a hotfix is tagged out of band (a v0.3.6 released after v0.4.0 would
|
||||
# sort as "newest" by date and silently downgrade the box).
|
||||
_target="$(git tag -l 'v*' --sort=-version:refname | head -1)"
|
||||
#
|
||||
# Filter to PLAIN vX.Y.Z: git's version sort ranks `v0.5.0-rc1` ABOVE `v0.5.0`
|
||||
# (verified), so without this a release candidate would be installed as though
|
||||
# it were the newest release. The release workflow deliberately supports
|
||||
# rc/beta/alpha tags, so they will exist.
|
||||
_target="$(git tag -l 'v*' --sort=-version:refname \
|
||||
| grep -E '^v[0-9]+\.[0-9]+\.[0-9]+$' | head -1)"
|
||||
if [ -z "$_target" ]; then
|
||||
echo "ERROR: no release tags found; use --main to track unreleased code." >&2
|
||||
exit 1
|
||||
@@ -149,6 +224,8 @@ if [ "$_current" = "$_target_sha" ]; then
|
||||
exit 0
|
||||
fi
|
||||
|
||||
_abort_if_untracked_blockers "$_target_sha"
|
||||
|
||||
# ── Show what is coming ───────────────────────────────────────────────────────
|
||||
|
||||
echo "Commits you do not have yet:"
|
||||
@@ -217,6 +294,12 @@ echo ""
|
||||
echo "=== Update complete ==="
|
||||
echo " $_current_desc -> $(git describe --tags --always)"
|
||||
echo ""
|
||||
if ! git symbolic-ref -q HEAD >/dev/null; then
|
||||
echo "NOTE: the checkout is now pinned to a release tag (detached HEAD), which is"
|
||||
echo " what you want for a deployment. Plain \`git pull\` will not work here —"
|
||||
echo " use \`bash update.sh\` from now on."
|
||||
echo ""
|
||||
fi
|
||||
echo "If anything looks wrong:"
|
||||
echo " bash $PATCH_DIR/update.sh --rollback # back to $_current_desc"
|
||||
echo " bash $PATCH_DIR/recover.sh # kill switch + restart"
|
||||
|
||||
Reference in New Issue
Block a user