Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7b4722d3f2 | ||
|
|
444932c369 | ||
|
|
8f2a6d3163 | ||
|
|
b98fc94179 | ||
|
|
762ee160c3 | ||
|
|
759579fc30 | ||
|
|
0e416176c6 | ||
|
|
e856cba6d7 | ||
|
|
e4d5f603d8 | ||
|
|
0a67699db4 | ||
|
|
f9b9cea84f | ||
|
|
4b6bce9b74 | ||
|
|
c3bf81d490 | ||
|
|
4529a63082 | ||
|
|
ec9d2ecd0f | ||
|
|
1df7c99c2e |
+1
-1
@@ -33,7 +33,7 @@ FORCE_CPU=false
|
||||
ENABLE_CACHE=true
|
||||
CACHE_DIR=/app/.if_cache
|
||||
HF_HOME=/models/huggingface
|
||||
INSIGHTFACE_HOME=/models
|
||||
INSIGHTFACE_HOME=/models/.insightface
|
||||
|
||||
# ── Tracker overrides (one-shot — remove after use) ───────────────────────────
|
||||
# DRY_RUN=true # Preview selection without downloading/uploading
|
||||
|
||||
@@ -15,22 +15,8 @@ env:
|
||||
IMAGE_NAME: sudolulo/winnow
|
||||
|
||||
jobs:
|
||||
verify-lockfile:
|
||||
name: Verify uv.lock is current
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v4
|
||||
|
||||
- name: Check lockfile is up to date
|
||||
run: uv lock --check
|
||||
|
||||
build:
|
||||
name: Build (${{ matrix.platform }})
|
||||
needs: verify-lockfile
|
||||
runs-on: ${{ matrix.runner }}
|
||||
strategy:
|
||||
matrix:
|
||||
@@ -140,4 +126,3 @@ jobs:
|
||||
- name: Inspect image
|
||||
run: |
|
||||
docker buildx imagetools inspect ${{ steps.tags.outputs.tags }}
|
||||
|
||||
|
||||
@@ -9,6 +9,8 @@ on:
|
||||
jobs:
|
||||
lint:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
env:
|
||||
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true
|
||||
steps:
|
||||
|
||||
@@ -14,22 +14,8 @@ concurrency:
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
verify-lockfile:
|
||||
name: Verify uv.lock is current
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v4
|
||||
|
||||
- name: Check lockfile is up to date
|
||||
run: uv lock --check
|
||||
|
||||
release:
|
||||
name: Create GitHub Release & Build Image
|
||||
needs: verify-lockfile
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: write
|
||||
@@ -48,6 +34,12 @@ jobs:
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v4
|
||||
|
||||
- name: Ensure uv.lock is current
|
||||
run: uv lock
|
||||
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v3
|
||||
|
||||
@@ -138,4 +130,3 @@ jobs:
|
||||
tags: |
|
||||
ghcr.io/sudolulo/winnow:latest
|
||||
ghcr.io/sudolulo/winnow:${{ steps.tag.outputs.TAG }}
|
||||
|
||||
|
||||
@@ -9,6 +9,8 @@ on:
|
||||
jobs:
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
|
||||
@@ -236,3 +236,4 @@ __marimo__/
|
||||
|
||||
# Streamlit
|
||||
.streamlit/secrets.toml
|
||||
compose.override.yml
|
||||
|
||||
+39
-3
@@ -7,6 +7,42 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
## [0.2.3] - 2026-06-12
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Clear-text API key storage**: `config.py` no longer writes `API_KEY` to `.immich_config.json`. The key must come from an environment variable or `.env` file. Interactive mode now prints a tip directing users to `.env`. Resolves CodeQL `py/clear-text-storage-sensitive-data`.
|
||||
|
||||
### Changed
|
||||
|
||||
- **CI workflow permissions**: `test.yml` and `lint.yml` now declare `permissions: contents: read`, following least-privilege principle and resolving `actions/missing-workflow-permissions` scanner alerts.
|
||||
- **CI lockfile race condition**: removed the `verify-lockfile` pre-job from `docker-publish.yml` and `release.yml`. The `update-lockfile.yml` bot maintains the lockfile; the verify step raced against it on the same push event and caused false failures. `release.yml` now runs `uv lock` inline so tag-triggered builds are always self-consistent.
|
||||
- **Docs moved to wiki**: `docs/` folder removed from the repository. Setup, Troubleshooting, and FAQ pages are now at the [GitHub wiki](https://github.com/sudolulo/winnow/wiki).
|
||||
|
||||
## [0.2.2] - 2026-06-12
|
||||
|
||||
### Added
|
||||
|
||||
- **Confidence scores in upload tracker**: Immich face confidence scores are now stored per asset in `frigate_uploaded_ids.json` under `by_person[name].scores`. Lays the groundwork for future replacement logic (remove low-confidence uploads when better images are found).
|
||||
- **Frigate-authoritative capacity tracking**: At startup, `GET /api/faces` is queried on the Frigate host to retrieve the actual number of trained images per person from the `train` directory (pending/unclassified queue is excluded). This count is stored as `frigate_count` in the tracker JSON so it survives Frigate downtime.
|
||||
- **Lifetime cap uses Frigate count**: `MAX_AUTO_IMAGES` is now enforced against Frigate's live training image count rather than the local uploaded-asset tally. Fallback priority: live Frigate API → last cached `frigate_count` in JSON → local uploaded count.
|
||||
- **Startup summary shows Frigate count**: Tracker summary at startup now includes the last known Frigate training count per person (e.g. `78 uploaded, 2 rejected, 42 in Frigate`).
|
||||
- **`winnow/frigate_api.py`**: new module encapsulating Frigate API helpers; currently exposes `get_frigate_face_counts()`.
|
||||
|
||||
### Changed
|
||||
|
||||
- `upload_tracker.py`: `by_person` entries migrated from flat list to `{asset_ids, scores, frigate_count}` dict. Old list format is read and migrated transparently on first write.
|
||||
- `mark_uploaded()` now accepts an optional `score` keyword argument.
|
||||
- `get_person_summary()` now returns `frigate_count` and `scores` fields alongside `uploaded` and `rejected`.
|
||||
|
||||
## [0.2.1] - 2026-06-12
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Container startup reinstalling packages**: `entrypoint.sh` used `uv run`, which performs a sync check on every startup and re-downloaded `ruff` and rebuilt the package each time. Replaced with direct `.venv/bin/python` calls to skip the sync entirely.
|
||||
- **InsightFace double `models/` path**: `INSIGHTFACE_HOME=/models` caused InsightFace to download Buffalo_L to `/models/models/buffalo_l` (InsightFace always appends `models/` to the root). Updated default in `compose.yml` and `.env.example` to `/models/.insightface`.
|
||||
- **Lint errors in CI**: unused imports in `tests/test_config.py` and `tests/test_upload_tracker.py`, unsorted imports in `scheduler.py` — all would have failed the ruff CI check.
|
||||
|
||||
## [0.2.0] - 2026-06-12
|
||||
|
||||
First release of winnow. Forked from [if-curator](https://github.com/ds-sebastian/if_curator) by Sebastian and rewritten for headless Docker deployment.
|
||||
@@ -21,14 +57,14 @@ First release of winnow. Forked from [if-curator](https://github.com/ds-sebastia
|
||||
|
||||
**Docker and scheduling**
|
||||
- `Dockerfile` — multi-stage build (CUDA 12.9 on amd64, plain Ubuntu on arm64); runtime stage excludes build tools (g++, python3.12-dev, curl, gnupg)
|
||||
- `compose.yml` — fully annotated with inline comments grouped by concern; TrueNAS volume paths in the example
|
||||
- `compose.yml` — fully annotated with inline comments grouped by concern
|
||||
- `entrypoint.sh` — runs the tool once on startup, then hands off to the scheduler if `CRON_SCHEDULE` is set
|
||||
- `scheduler.py` — in-process cron scheduler that keeps the container (and loaded models) alive between runs
|
||||
- `CRON_SCHEDULE` env var — standard cron expression for recurring runs; unset exits after first run
|
||||
- `.dockerignore` — keeps `.venv`, `__pycache__`, test files, and logs out of the image context
|
||||
- Multi-arch image: `linux/amd64` and `linux/arm64` built and merged into a single manifest on GHCR
|
||||
- `tini` as PID 1 init process for correct signal handling
|
||||
- Non-root container user (`appuser`, uid 568) matching TrueNAS default app UID
|
||||
- Non-root container user (`appuser`, uid 568)
|
||||
- `HEALTHCHECK` in Dockerfile
|
||||
|
||||
**Object mode**
|
||||
@@ -73,7 +109,7 @@ First release of winnow. Forked from [if-curator](https://github.com/ds-sebastia
|
||||
- 24 unit tests across four modules: `test_config`, `test_immich_api`, `test_jobs`, `test_upload_tracker`
|
||||
|
||||
**Documentation**
|
||||
- `docs/setup.md` — step-by-step install guide with TrueNAS and bare-Docker sections
|
||||
- `docs/setup.md` — step-by-step install and GPU passthrough guide
|
||||
- `docs/troubleshooting.md` — common failure modes with fixes
|
||||
- `docs/faq.md` — answers to questions new users will ask
|
||||
- `.env.example` — copy-paste starting point with every env var and inline comments
|
||||
|
||||
@@ -68,6 +68,7 @@ RUN groupadd -g 568 apps && useradd -u 568 -g apps -m -s /bin/bash appuser \
|
||||
&& mkdir -p /models/.insightface /models/huggingface \
|
||||
&& chown -R appuser:apps /app /models
|
||||
|
||||
WORKDIR /app
|
||||
USER appuser
|
||||
ENV HF_HOME=/models/huggingface INSIGHTFACE_HOME=/models
|
||||
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
[](https://github.com/sudolulo/winnow/actions/workflows/docker-publish.yml) [](https://github.com/sudolulo/winnow/actions/workflows/release.yml) [](https://github.com/sudolulo/winnow/actions/workflows/lint.yml) [](https://github.com/sudolulo/winnow/actions/workflows/test.yml)
|
||||
[](https://immich.app) [](https://frigate.video)
|
||||
|
||||
**Docs:** [Setup Guide](docs/setup.md) · [Troubleshooting](docs/troubleshooting.md) · [FAQ](docs/faq.md)
|
||||
|
||||
@@ -16,6 +17,8 @@ Frigate's face recognition model (ArcFace) and object classifier are only as goo
|
||||
|
||||
If you upload 100 photos from the same week, the model learns the lighting in your living room and the jacket you wore that month. It struggles the moment anything changes. What you actually want is a spread: different years, different lighting conditions, different angles, different contexts.
|
||||
|
||||
This is especially true for people who have never been to your property, or who visit rarely — family members, friends, anyone Frigate has never seen in person. Live detections alone will never build a reliable model for these people. Your photo library already has the data; winnow finds and delivers the right subset of it.
|
||||
|
||||
Finding that spread manually across a library of thousands of photos is not practical. `winnow` does it automatically.
|
||||
|
||||
---
|
||||
@@ -68,6 +71,14 @@ Uploaded asset IDs are recorded so the same image is never uploaded twice, even
|
||||
|
||||
---
|
||||
|
||||
## Note on Crop Quality
|
||||
|
||||
winnow works well, but no automated pipeline is perfect. Occasionally a bad crop will slip through quality filtering — a partial face, someone in the background, a blurry frame. After a run it's worth a quick review in Frigate's face management UI to remove anything that doesn't belong.
|
||||
|
||||
Issues and feedback welcome via [GitHub Issues](https://github.com/sudolulo/winnow/issues).
|
||||
|
||||
---
|
||||
|
||||
## Modes
|
||||
|
||||
### Face Mode (default)
|
||||
@@ -126,7 +137,7 @@ services:
|
||||
capabilities: [gpu]
|
||||
```
|
||||
|
||||
See [compose.yml](compose.yml) for the full annotated example including TrueNAS volume paths.
|
||||
See [compose.yml](compose.yml) for the full annotated example.
|
||||
|
||||
### Scheduling Behaviour
|
||||
|
||||
@@ -227,4 +238,4 @@ Requires Python 3.12+ and [uv](https://astral.sh/uv/). An NVIDIA GPU is strongly
|
||||
|
||||
## Attribution
|
||||
|
||||
Based on [winnow](https://github.com/ds-sebastian/if_curator) by Sebastian, licensed MIT.
|
||||
Based on [if_curator](https://github.com/ds-sebastian/if_curator) by Sebastian, licensed MIT.
|
||||
|
||||
@@ -1,8 +0,0 @@
|
||||
services:
|
||||
winnow:
|
||||
build: .
|
||||
network_mode: host
|
||||
volumes:
|
||||
- /code/winnow/models:/models
|
||||
- /code/winnow/embeddings:/app/.if_cache
|
||||
- /code/winnow/output:/app/frigate_train
|
||||
+5
-5
@@ -38,7 +38,7 @@ services:
|
||||
- ENABLE_CACHE=true
|
||||
- CACHE_DIR=/app/.if_cache
|
||||
- HF_HOME=/models/huggingface
|
||||
- INSIGHTFACE_HOME=/models
|
||||
- INSIGHTFACE_HOME=/models/.insightface
|
||||
|
||||
# ── Tracker overrides (one-shot, remove after use) ────────────────────
|
||||
# - DRY_RUN=true # Preview selection without downloading/uploading
|
||||
@@ -52,10 +52,10 @@ services:
|
||||
# - CRON_SCHEDULE=0 3 1 * *
|
||||
# - CRON_SCHEDULE=*/30 * * * *
|
||||
volumes:
|
||||
# Replace <pool> with your TrueNAS pool name, e.g. /mnt/tank/winnow/...
|
||||
- /mnt/<pool>/winnow/models:/models
|
||||
- /mnt/<pool>/winnow/embeddings:/app/.if_cache
|
||||
- /mnt/<pool>/winnow/output:/app/frigate_train
|
||||
# Replace with absolute paths on your host, e.g. /opt/winnow/models
|
||||
- /path/to/winnow/models:/models
|
||||
- /path/to/winnow/cache:/app/.if_cache
|
||||
- /path/to/winnow/output:/app/frigate_train
|
||||
stdin_open: true
|
||||
tty: true
|
||||
restart: unless-stopped
|
||||
|
||||
-63
@@ -1,63 +0,0 @@
|
||||
# FAQ
|
||||
|
||||
## Does winnow modify my Immich library?
|
||||
|
||||
No. winnow only reads from Immich (assets, people, face bounding boxes). It never writes back to Immich or deletes anything.
|
||||
|
||||
---
|
||||
|
||||
## How many images should I upload to Frigate?
|
||||
|
||||
The `auto` strategy decides this for you — it keeps selecting until adding more images would be redundant. In practice this is usually 20–60 per person. You can cap it with `MAX_AUTO_IMAGES` (default 80).
|
||||
|
||||
Quality and diversity matter far more than volume. 30 well-spread images outperform 200 from the same week.
|
||||
|
||||
---
|
||||
|
||||
## What's the difference between face mode and object mode?
|
||||
|
||||
- **Face mode**: Extracts and aligns face crops, uploads them directly to Frigate's face training API. This is for teaching Frigate to recognize specific people.
|
||||
- **Object mode**: Runs YOLO detection on full images and saves crops of a target class (dog, cat, car, etc.) to disk. Frigate has no API for object training data, so you place them manually.
|
||||
|
||||
---
|
||||
|
||||
## Can I run it without Frigate?
|
||||
|
||||
Yes — in object mode, `FRIGATE_URL` is not used and crops are saved to the output volume. In face mode you need Frigate to receive the uploads, but you can use `DRY_RUN=true` to preview selection without uploading.
|
||||
|
||||
---
|
||||
|
||||
## How does auto-diversity mode work?
|
||||
|
||||
winnow computes a vector embedding for each candidate image (what the face/object actually looks like — angle, lighting, expression). It then clusters those embeddings and picks representatives that are maximally spread across the embedding space. It stops when the next-most-different image is already close to something already selected. See the README for the full pipeline.
|
||||
|
||||
---
|
||||
|
||||
## Does it support multiple people in one run?
|
||||
|
||||
Yes. By default it processes every named person in your Immich library. Use `ONLY_PEOPLE` to whitelist specific names or `SKIP_PEOPLE` to exclude them.
|
||||
|
||||
---
|
||||
|
||||
## What GPU is needed?
|
||||
|
||||
Any NVIDIA GPU with CUDA 12.x support. The models (InsightFace Buffalo_L + SigLIP) fit comfortably in 4 GB VRAM. CPU mode works but is significantly slower.
|
||||
|
||||
ARM builds (linux/arm64) use CPU-only — CUDA is not available on ARM.
|
||||
|
||||
---
|
||||
|
||||
## Does it work on Unraid / Proxmox / bare Docker?
|
||||
|
||||
Yes — the `compose.yml` uses standard Docker volume mounts. The TrueNAS paths in the example (`/mnt/<pool>/...`) are just an example; replace them with whatever paths suit your setup.
|
||||
|
||||
---
|
||||
|
||||
## How do I update winnow?
|
||||
|
||||
```bash
|
||||
docker compose pull
|
||||
docker compose up -d
|
||||
```
|
||||
|
||||
The `latest` tag on GHCR tracks the `main` branch. Pinning to a version tag (e.g. `ghcr.io/sudolulo/winnow:v0.2.0`) is recommended for stability.
|
||||
@@ -1,88 +0,0 @@
|
||||
# Setup Guide
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- [Immich](https://immich.app) v1.106+ with face recognition enabled and people tagged
|
||||
- [Frigate](https://frigate.video) v0.16+ (face mode only)
|
||||
- Docker with the NVIDIA container toolkit (optional but strongly recommended)
|
||||
|
||||
---
|
||||
|
||||
## 1. Get your Immich API key
|
||||
|
||||
1. Open Immich → **Account Settings** → **API Keys**
|
||||
2. Click **New API Key**, give it a name (e.g. `winnow`), copy the key
|
||||
|
||||
---
|
||||
|
||||
## 2. Get your Frigate URL
|
||||
|
||||
This is the base URL of your Frigate instance, e.g. `http://192.168.1.10:5000`. Only needed for face mode — omit it entirely if you're using object mode.
|
||||
|
||||
---
|
||||
|
||||
## 3. Deploy with Docker Compose
|
||||
|
||||
Copy [`compose.yml`](../compose.yml) and [`.env.example`](../.env.example) to a directory on your host:
|
||||
|
||||
```bash
|
||||
mkdir winnow && cd winnow
|
||||
curl -O https://raw.githubusercontent.com/sudolulo/winnow/main/compose.yml
|
||||
curl -O https://raw.githubusercontent.com/sudolulo/winnow/main/.env.example
|
||||
cp .env.example .env
|
||||
```
|
||||
|
||||
Edit `.env` with your values:
|
||||
|
||||
```bash
|
||||
IMMICH_URL=http://192.168.1.10:2283
|
||||
API_KEY=your-immich-api-key
|
||||
FRIGATE_URL=http://192.168.1.10:5000
|
||||
```
|
||||
|
||||
Edit the volume paths in `compose.yml` to match your storage layout (replace `/mnt/<pool>` with your actual path).
|
||||
|
||||
Start it:
|
||||
|
||||
```bash
|
||||
docker compose up -d
|
||||
```
|
||||
|
||||
Logs:
|
||||
|
||||
```bash
|
||||
docker compose logs -f winnow
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 4. First run
|
||||
|
||||
On the first run, winnow downloads the embedding models (~1–2 GB) from HuggingFace and InsightFace. This happens once — subsequent runs use the cached models from your mounted volume and start immediately.
|
||||
|
||||
---
|
||||
|
||||
## 5. Scheduling
|
||||
|
||||
Set `CRON_SCHEDULE` in your `.env` to keep winnow running on a schedule:
|
||||
|
||||
```
|
||||
CRON_SCHEDULE=0 3 * * 0 # Every Sunday at 3 AM
|
||||
```
|
||||
|
||||
Without `CRON_SCHEDULE`, the container runs once and exits.
|
||||
|
||||
---
|
||||
|
||||
## TrueNAS Scale
|
||||
|
||||
The included `compose.yml` uses TrueNAS-style volume paths. Replace `<pool>` with your pool name:
|
||||
|
||||
```yaml
|
||||
volumes:
|
||||
- /mnt/tank/winnow/models:/models
|
||||
- /mnt/tank/winnow/embeddings:/app/.if_cache
|
||||
- /mnt/tank/winnow/output:/app/frigate_train
|
||||
```
|
||||
|
||||
GPU passthrough on TrueNAS requires the NVIDIA app to be installed from the TrueNAS catalog and the `deploy.resources.reservations.devices` block in `compose.yml` (already included).
|
||||
@@ -1,77 +0,0 @@
|
||||
# Troubleshooting
|
||||
|
||||
## Container exits immediately
|
||||
|
||||
Check logs:
|
||||
```bash
|
||||
docker compose logs winnow
|
||||
```
|
||||
|
||||
Common causes:
|
||||
- **Missing required env var** — `IMMICH_URL` or `API_KEY` not set
|
||||
- **Cannot reach Immich** — check the URL and that Immich is running; use `http://` not `https://` unless you have TLS set up
|
||||
|
||||
---
|
||||
|
||||
## "No people found" / nothing processed
|
||||
|
||||
- Make sure Immich has completed face recognition and you have named people in your library
|
||||
- `YEARS_FILTER` defaults to 10 years — increase it if your tagged photos are older
|
||||
- `MIN_FACE_COUNT` skips people with few photos — lower or remove it
|
||||
|
||||
---
|
||||
|
||||
## Frigate upload fails
|
||||
|
||||
- Confirm `FRIGATE_URL` is reachable from inside the container: `docker exec winnow curl $FRIGATE_URL/api/stats`
|
||||
- Check Frigate v0.16+ — older versions don't have the face training API
|
||||
- Set `DRY_RUN=true` to verify selection without uploading
|
||||
|
||||
---
|
||||
|
||||
## Models fail to download
|
||||
|
||||
winnow downloads InsightFace and HuggingFace (SigLIP) models on first run.
|
||||
|
||||
- Ensure the container has internet access
|
||||
- Confirm the model volume is mounted and writable
|
||||
- If behind a proxy, set `HTTP_PROXY` / `HTTPS_PROXY` env vars
|
||||
|
||||
---
|
||||
|
||||
## Running on CPU (no GPU)
|
||||
|
||||
Set `FORCE_CPU=true`. Everything works but embedding computation is slower — expect several minutes per person instead of seconds.
|
||||
|
||||
If you have a GPU but it's not being used:
|
||||
- Confirm the NVIDIA container toolkit is installed: `docker run --rm --gpus all nvidia/cuda:12.9.2-base-ubuntu22.04 nvidia-smi`
|
||||
- Confirm the `deploy.resources.reservations.devices` block is present in `compose.yml`
|
||||
|
||||
---
|
||||
|
||||
## Same images uploaded every run
|
||||
|
||||
The upload tracker is stored in `CACHE_DIR` (`/app/.if_cache` by default). If this volume isn't persisted between runs, the tracker resets and images are re-uploaded.
|
||||
|
||||
Make sure `/app/.if_cache` is mounted to a persistent host path.
|
||||
|
||||
---
|
||||
|
||||
## Re-uploading a specific person
|
||||
|
||||
To clear the upload history for one person and start fresh:
|
||||
|
||||
```env
|
||||
RESET_PERSON=John
|
||||
```
|
||||
|
||||
Remove this after one run — it clears the history and then processes normally.
|
||||
|
||||
---
|
||||
|
||||
## Image quality issues
|
||||
|
||||
- **Too blurry**: Lower `BLUR_THRESHOLD` (default 100) — e.g. `50` accepts more blur
|
||||
- **Face too small**: Lower `MIN_FACE_WIDTH` (default 50px)
|
||||
- **Low confidence detections included**: Raise `MIN_CONFIDENCE` (default 0.7)
|
||||
- **Rejected images being re-tried**: Set `RETRY_REJECTED=true` for one run
|
||||
+2
-2
@@ -4,13 +4,13 @@ export PYTHONUNBUFFERED=1
|
||||
|
||||
# 1. Run the job immediately on startup
|
||||
echo "▶ Running on startup..."
|
||||
uv run python -m winnow.cli
|
||||
/app/.venv/bin/python -m winnow.cli
|
||||
|
||||
# 2. If a schedule exists, start the scheduler
|
||||
if [ -n "${CRON_SCHEDULE:-}" ]; then
|
||||
echo "▶ CRON_SCHEDULE set to: $CRON_SCHEDULE"
|
||||
echo "▶ Switching to scheduled mode..."
|
||||
exec uv run python3 /app/scheduler.py
|
||||
exec /app/.venv/bin/python /app/scheduler.py
|
||||
else
|
||||
echo "▶ No schedule set, exiting."
|
||||
fi
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[project]
|
||||
name = "winnow"
|
||||
version = "0.2.0"
|
||||
version = "0.2.3"
|
||||
description = "Immich to Frigate training sets"
|
||||
license = "MIT"
|
||||
requires-python = ">=3.12"
|
||||
|
||||
+4
-4
@@ -1,9 +1,9 @@
|
||||
#!/usr/bin/env python3
|
||||
import os
|
||||
import sys
|
||||
import subprocess
|
||||
import time
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
|
||||
@@ -1,8 +1,6 @@
|
||||
"""Smoke tests for configuration loading."""
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
def test_config_loads_defaults(monkeypatch):
|
||||
|
||||
@@ -1,8 +1,5 @@
|
||||
"""Tests for upload tracker — mark, filter, reset, and summary logic."""
|
||||
|
||||
import json
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
@@ -2289,7 +2289,7 @@ wheels = [
|
||||
|
||||
[[package]]
|
||||
name = "winnow"
|
||||
version = "0.2.0"
|
||||
version = "0.2.3"
|
||||
source = { editable = "." }
|
||||
dependencies = [
|
||||
{ name = "croniter" },
|
||||
|
||||
+9
-1
@@ -48,7 +48,15 @@ def main() -> None:
|
||||
if summary:
|
||||
rprint("\n[dim]Tracker summary:[/dim]")
|
||||
for person_name, counts in summary.items():
|
||||
rprint(f" [dim]{person_name}: {counts['uploaded']} uploaded, {counts['rejected']} rejected[/dim]")
|
||||
frigate_part = (
|
||||
f", {counts['frigate_count']} in Frigate"
|
||||
if counts.get("frigate_count") is not None
|
||||
else ""
|
||||
)
|
||||
rprint(
|
||||
f" [dim]{person_name}: {counts['uploaded']} uploaded,"
|
||||
f" {counts['rejected']} rejected{frigate_part}[/dim]"
|
||||
)
|
||||
|
||||
people = get_people()
|
||||
if not people:
|
||||
|
||||
+7
-5
@@ -67,12 +67,11 @@ class _Config:
|
||||
self.ENABLE_CACHE = os.getenv("ENABLE_CACHE", "false").lower() in ("true", "1", "yes")
|
||||
self.CACHE_DIR = os.getenv("CACHE_DIR", ".if_cache")
|
||||
|
||||
# Fall back to config file for missing values
|
||||
# Fall back to config file for non-sensitive values (API_KEY not stored here)
|
||||
if CONFIG_FILE.exists():
|
||||
try:
|
||||
data = json.loads(CONFIG_FILE.read_text())
|
||||
self.IMMICH_URL = self.IMMICH_URL or data.get("IMMICH_URL")
|
||||
self.API_KEY = self.API_KEY or data.get("API_KEY")
|
||||
if not os.getenv("OUTPUT_DIR"):
|
||||
self.OUTPUT_DIR = data.get("OUTPUT_DIR", self.OUTPUT_DIR)
|
||||
except (json.JSONDecodeError, OSError) as e:
|
||||
@@ -84,13 +83,16 @@ class _Config:
|
||||
cls._instance = None
|
||||
|
||||
def save(self) -> None:
|
||||
"""Persist configuration to file."""
|
||||
"""Persist non-sensitive configuration to file.
|
||||
|
||||
API_KEY is intentionally excluded — store it in .env or as an
|
||||
environment variable instead of a plain-text config file.
|
||||
"""
|
||||
try:
|
||||
CONFIG_FILE.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"IMMICH_URL": self.IMMICH_URL,
|
||||
"API_KEY": self.API_KEY,
|
||||
"OUTPUT_DIR": self.OUTPUT_DIR,
|
||||
},
|
||||
indent=2,
|
||||
@@ -113,8 +115,8 @@ class _Config:
|
||||
|
||||
if not self.API_KEY:
|
||||
console.print("[yellow]Immich API Key not found.[/yellow]")
|
||||
console.print("[dim]Tip: set API_KEY in your .env file to avoid re-entering it.[/dim]")
|
||||
self.API_KEY = Prompt.ask("Enter Immich API Key", password=True)
|
||||
self.save()
|
||||
|
||||
def validate(self) -> None:
|
||||
"""Raise ValueError if required config is missing."""
|
||||
|
||||
+9
-3
@@ -58,6 +58,7 @@ def _enrich_asset_with_face_data(asset: dict, person: dict) -> dict:
|
||||
|
||||
# Inject into asset so process_face_mode can find it via asset["people"]
|
||||
asset["people"] = [{"id": person_id, "faces": [face_info]}]
|
||||
asset["face_confidence"] = face_data.confidence
|
||||
return asset
|
||||
|
||||
|
||||
@@ -96,8 +97,9 @@ def execute_jobs(jobs: list[dict]) -> None:
|
||||
shutil.rmtree(person_dir)
|
||||
os.makedirs(person_dir, exist_ok=True)
|
||||
|
||||
# Track filename → asset_id mapping for upload dedup
|
||||
# Track filename → asset_id and filename → confidence score
|
||||
asset_map: dict[str, str] = {}
|
||||
score_map: dict[str, float | None] = {}
|
||||
|
||||
count = 0
|
||||
for asset in assets:
|
||||
@@ -132,11 +134,13 @@ def execute_jobs(jobs: list[dict]) -> None:
|
||||
# Record which asset produced which output file
|
||||
filename = f"{count}.jpg"
|
||||
asset_map[filename] = asset["id"]
|
||||
score_map[filename] = asset.get("face_confidence")
|
||||
# Also record object-mode variant filenames
|
||||
if mode == "object":
|
||||
for f in sorted(os.listdir(person_dir)):
|
||||
if f.startswith(f"{count}_") and f not in asset_map:
|
||||
asset_map[f] = asset["id"]
|
||||
score_map[f] = asset.get("face_confidence")
|
||||
|
||||
count += 1
|
||||
else:
|
||||
@@ -149,8 +153,9 @@ def execute_jobs(jobs: list[dict]) -> None:
|
||||
progress.advance(job_task)
|
||||
progress.advance(overall_task)
|
||||
|
||||
# Store asset_map on the job so upload_to_frigate can use it
|
||||
# Store maps on the job so upload_to_frigate can use them
|
||||
job["asset_map"] = asset_map
|
||||
job["score_map"] = score_map
|
||||
|
||||
progress.remove_task(job_task)
|
||||
|
||||
@@ -230,6 +235,7 @@ def upload_to_frigate(jobs: list[dict]) -> None:
|
||||
continue
|
||||
|
||||
asset_map = filename_to_asset_id.get(name, {})
|
||||
score_map = job.get("score_map", {})
|
||||
person_files = sorted(asset_map.keys())
|
||||
|
||||
if not person_files:
|
||||
@@ -257,7 +263,7 @@ def upload_to_frigate(jobs: list[dict]) -> None:
|
||||
# Mark this asset as uploaded so it's skipped on future runs
|
||||
asset_id = asset_map.get(fname)
|
||||
if asset_id:
|
||||
mark_uploaded(asset_id, person_name=name)
|
||||
mark_uploaded(asset_id, person_name=name, score=score_map.get(fname))
|
||||
|
||||
break
|
||||
else:
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
"""Frigate API helpers for querying face training state."""
|
||||
|
||||
import logging
|
||||
import os
|
||||
|
||||
import requests
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def get_frigate_face_counts() -> dict[str, int] | None:
|
||||
"""Return {person_name: training_image_count} from Frigate's train directory.
|
||||
|
||||
Returns None if FRIGATE_URL is not set or the API is unreachable, so callers
|
||||
can distinguish "API unavailable" from "person has 0 images."
|
||||
"""
|
||||
frigate_url = os.environ.get("FRIGATE_URL", "").rstrip("/")
|
||||
if not frigate_url:
|
||||
return None
|
||||
try:
|
||||
resp = requests.get(f"{frigate_url}/api/faces", timeout=10)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
train = data.get("train", {})
|
||||
return {name: len(files) for name, files in train.items() if isinstance(files, list)}
|
||||
except Exception as e:
|
||||
logger.warning(f"Could not query Frigate face counts: {e}")
|
||||
return None
|
||||
+33
-1
@@ -11,9 +11,10 @@ from rich.table import Table
|
||||
from .config import Config
|
||||
from .diversity import select_diverse_assets
|
||||
from .embeddings import is_embedding_available, load_embedding_model
|
||||
from .frigate_api import get_frigate_face_counts
|
||||
from .immich_api import fetch_all_assets, filter_recent_assets
|
||||
from .logging import console
|
||||
from .upload_tracker import filter_already_uploaded
|
||||
from .upload_tracker import filter_already_uploaded, get_person_summary, update_frigate_count
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -236,6 +237,12 @@ def auto_configure(people: list[dict]) -> list[dict]:
|
||||
f" ≥{min_face_count} assets (MIN_FACE_COUNT={min_face_count})"
|
||||
)
|
||||
|
||||
frigate_counts = get_frigate_face_counts()
|
||||
# Persist each count to tracker so the last known value survives Frigate downtime
|
||||
if frigate_counts is not None:
|
||||
for pname, count in frigate_counts.items():
|
||||
update_frigate_count(pname, count)
|
||||
upload_summary = get_person_summary()
|
||||
jobs = []
|
||||
for person in valid_people:
|
||||
name = person["name"]
|
||||
@@ -263,9 +270,34 @@ def auto_configure(people: list[dict]) -> list[dict]:
|
||||
rprint(f" [dim]Skipping {name} (0 new images after dedup).[/dim]")
|
||||
continue
|
||||
|
||||
# Enforce MAX_AUTO_IMAGES as a lifetime cap per person.
|
||||
# Priority: live Frigate count → last cached Frigate count → local uploaded count.
|
||||
person_summary = upload_summary.get(name, {})
|
||||
if frigate_counts is not None:
|
||||
already_uploaded = frigate_counts.get(name, 0)
|
||||
else:
|
||||
already_uploaded = (
|
||||
person_summary.get("frigate_count")
|
||||
or person_summary.get("uploaded", 0)
|
||||
)
|
||||
capacity = Config.MAX_AUTO_IMAGES - already_uploaded
|
||||
if capacity <= 0:
|
||||
rprint(
|
||||
f" [dim]Skipping {name} (at lifetime cap:"
|
||||
f" {already_uploaded}/{Config.MAX_AUTO_IMAGES} trained).[/dim]"
|
||||
)
|
||||
continue
|
||||
|
||||
has_embedding = is_embedding_available(entity_type)
|
||||
limit, selection_mode = _resolve_strategy(strategy, has_embedding)
|
||||
|
||||
# Cap selection to remaining capacity
|
||||
if limit == "auto":
|
||||
if already_uploaded > 0:
|
||||
limit = capacity # partially filled — select exactly what remains
|
||||
else:
|
||||
limit = min(limit, capacity)
|
||||
|
||||
if selection_mode == "skip":
|
||||
continue
|
||||
|
||||
|
||||
+61
-19
@@ -8,6 +8,13 @@ Both are excluded from future candidate pools. To reset:
|
||||
- All: delete both files
|
||||
- One person: call reset_person("Name") or set RESET_PERSON=Name
|
||||
- Rejects only: delete frigate_rejected_ids.json, or set RETRY_REJECTED=true
|
||||
|
||||
by_person schema (frigate_uploaded_ids.json):
|
||||
{
|
||||
"asset_ids": ["immich-id-1", ...], # all assets we attempted to upload
|
||||
"scores": {"immich-id-1": 0.953}, # Immich face confidence at upload time
|
||||
"frigate_count": 42 # last known Frigate training image count
|
||||
}
|
||||
"""
|
||||
|
||||
import json
|
||||
@@ -55,7 +62,23 @@ def _load_flat(filename: str) -> set[str]:
|
||||
return set(_load(filename).get(_flat_key(filename), []))
|
||||
|
||||
|
||||
def _mark(filename: str, asset_id: str, person_name: str | None) -> None:
|
||||
def _get_ids(entry: list | dict) -> list[str]:
|
||||
"""Extract asset_ids from either the old list format or the new dict format."""
|
||||
if isinstance(entry, list):
|
||||
return entry
|
||||
return entry.get("asset_ids", [])
|
||||
|
||||
|
||||
def _migrate_entry(entry: list | dict) -> dict:
|
||||
"""Ensure by_person entry is in the current dict format."""
|
||||
if isinstance(entry, list):
|
||||
return {"asset_ids": sorted(entry), "scores": {}}
|
||||
entry.setdefault("asset_ids", [])
|
||||
entry.setdefault("scores", {})
|
||||
return entry
|
||||
|
||||
|
||||
def _mark(filename: str, asset_id: str, person_name: str | None, score: float | None = None) -> None:
|
||||
data = _load(filename)
|
||||
flat_key = _flat_key(filename)
|
||||
flat = set(data.get(flat_key, []))
|
||||
@@ -63,9 +86,13 @@ def _mark(filename: str, asset_id: str, person_name: str | None) -> None:
|
||||
data[flat_key] = sorted(flat)
|
||||
if person_name:
|
||||
by_person = data.setdefault("by_person", {})
|
||||
person_ids = set(by_person.get(person_name, []))
|
||||
person_ids.add(asset_id)
|
||||
by_person[person_name] = sorted(person_ids)
|
||||
entry = _migrate_entry(by_person.get(person_name, {}))
|
||||
ids = set(entry["asset_ids"])
|
||||
ids.add(asset_id)
|
||||
entry["asset_ids"] = sorted(ids)
|
||||
if score is not None:
|
||||
entry["scores"][asset_id] = round(score, 4)
|
||||
by_person[person_name] = entry
|
||||
_save(filename, data)
|
||||
|
||||
|
||||
@@ -79,8 +106,8 @@ def load_rejected_ids() -> set[str]:
|
||||
return _load_flat(REJECT_TRACKER_FILE)
|
||||
|
||||
|
||||
def mark_uploaded(asset_id: str, person_name: str | None = None) -> None:
|
||||
_mark(UPLOAD_TRACKER_FILE, asset_id, person_name)
|
||||
def mark_uploaded(asset_id: str, person_name: str | None = None, score: float | None = None) -> None:
|
||||
_mark(UPLOAD_TRACKER_FILE, asset_id, person_name, score=score)
|
||||
logger.debug(f"Marked {asset_id} as uploaded ({person_name})")
|
||||
|
||||
|
||||
@@ -89,14 +116,25 @@ def mark_rejected(asset_id: str, person_name: str | None = None) -> None:
|
||||
logger.debug(f"Marked {asset_id} as rejected ({person_name})")
|
||||
|
||||
|
||||
def update_frigate_count(person_name: str, count: int) -> None:
|
||||
"""Record Frigate's authoritative training image count for a person."""
|
||||
data = _load(UPLOAD_TRACKER_FILE)
|
||||
by_person = data.setdefault("by_person", {})
|
||||
entry = _migrate_entry(by_person.get(person_name, {}))
|
||||
entry["frigate_count"] = count
|
||||
by_person[person_name] = entry
|
||||
_save(UPLOAD_TRACKER_FILE, data)
|
||||
|
||||
|
||||
def reset_person(person_name: str) -> None:
|
||||
"""Remove all uploaded and rejected records for a given person."""
|
||||
for filename in (UPLOAD_TRACKER_FILE, REJECT_TRACKER_FILE):
|
||||
data = _load(filename)
|
||||
flat_key = _flat_key(filename)
|
||||
by_person = data.get("by_person", {})
|
||||
person_ids = set(by_person.pop(person_name, []))
|
||||
if person_ids:
|
||||
entry = by_person.pop(person_name, None)
|
||||
if entry is not None:
|
||||
person_ids = set(_get_ids(entry))
|
||||
flat = set(data.get(flat_key, [])) - person_ids
|
||||
data[flat_key] = sorted(flat)
|
||||
data["by_person"] = by_person
|
||||
@@ -104,18 +142,22 @@ def reset_person(person_name: str) -> None:
|
||||
logger.info(f"Reset tracking data for {person_name}")
|
||||
|
||||
|
||||
def get_person_summary() -> dict[str, dict[str, int]]:
|
||||
"""Return {person_name: {uploaded: N, rejected: N}} for display."""
|
||||
uploaded_by = _load(UPLOAD_TRACKER_FILE).get("by_person", {})
|
||||
rejected_by = _load(REJECT_TRACKER_FILE).get("by_person", {})
|
||||
names = set(uploaded_by) | set(rejected_by)
|
||||
return {
|
||||
name: {
|
||||
"uploaded": len(uploaded_by.get(name, [])),
|
||||
"rejected": len(rejected_by.get(name, [])),
|
||||
def get_person_summary() -> dict[str, dict]:
|
||||
"""Return {person_name: {uploaded, rejected, frigate_count, scores}} for display/capacity."""
|
||||
uploaded_data = _load(UPLOAD_TRACKER_FILE).get("by_person", {})
|
||||
rejected_data = _load(REJECT_TRACKER_FILE).get("by_person", {})
|
||||
names = set(uploaded_data) | set(rejected_data)
|
||||
result = {}
|
||||
for name in sorted(names):
|
||||
u_entry = uploaded_data.get(name, {})
|
||||
r_entry = rejected_data.get(name, {})
|
||||
result[name] = {
|
||||
"uploaded": len(_get_ids(u_entry)),
|
||||
"rejected": len(_get_ids(r_entry)),
|
||||
"frigate_count": u_entry.get("frigate_count") if isinstance(u_entry, dict) else None,
|
||||
"scores": u_entry.get("scores", {}) if isinstance(u_entry, dict) else {},
|
||||
}
|
||||
for name in sorted(names)
|
||||
}
|
||||
return result
|
||||
|
||||
|
||||
def filter_already_uploaded(
|
||||
|
||||
Reference in New Issue
Block a user