mirror of
https://github.com/xCyanGrizzly/DragonsStash.git
synced 2026-09-21 05:21:43 +00:00
Compare commits
69
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
148d688d43 | ||
|
|
b28e38d233 | ||
|
|
662f5ac711 | ||
|
|
960da01ec6 | ||
|
|
a46e746298 | ||
|
|
26be615918 | ||
|
|
eda882dc90 | ||
|
|
80d41ac78f | ||
|
|
40894267d4 | ||
|
|
9a45fdf6d9 | ||
|
|
809d72660d | ||
|
|
6102fd474f | ||
|
|
dadf03212c | ||
|
|
267c72bbe8 | ||
|
|
497c4876a6 | ||
|
|
f5d913eb18 | ||
|
|
1e11dd3fd8 | ||
|
|
086f58f9dd | ||
|
|
3595f6f097 | ||
|
|
abdfa437d9 | ||
|
|
49f14bcb0d | ||
|
|
d4a1cfec99 | ||
|
|
ecedc0fec4 | ||
|
|
1b1f5b7972 | ||
|
|
e822ea3e76 | ||
|
|
9d16156161 | ||
|
|
2e7e6cca9b | ||
|
|
ceae4f384b | ||
|
|
7595543386 | ||
|
|
09ee9da9cc | ||
|
|
7ddf13053f | ||
|
|
4a7b9e2a09 | ||
|
|
9a06130c5b | ||
|
|
a7aa4ce285 | ||
|
|
38072d250f | ||
|
|
0bce1168a9 | ||
|
|
018b0f5d74 | ||
|
|
c2590fb66f | ||
|
|
8b443620c8 | ||
|
|
80aa2b0ee0 | ||
|
|
f0e0e79d34 | ||
|
|
21bd46010f | ||
|
|
2252ac01f5 | ||
|
|
63df348028 | ||
|
|
8bbf51f056 | ||
|
|
5873d148c1 | ||
|
|
90be365c7f | ||
|
|
7e21a41615 | ||
|
|
f3d62c68fb | ||
|
|
412e3066bc | ||
|
|
ffe5c920c6 | ||
|
|
c533d034f9 | ||
|
|
03822e0763 | ||
|
|
3536089d52 | ||
|
|
a8818dcf0c | ||
|
|
d57ec0458f | ||
|
|
6763af731f | ||
|
|
d5ba4fd4fd | ||
|
|
def8edb029 | ||
|
|
68742d682d | ||
|
|
f8bc737214 | ||
|
|
238ec17155 | ||
|
|
a2b7e77cc4 | ||
|
|
2beac6d62f | ||
|
|
897cd3d95e | ||
|
|
34e88a8cf5 | ||
|
|
5e7807b056 | ||
|
|
7693ed6f02 | ||
|
|
75c67b2036 |
@@ -89,7 +89,14 @@
|
||||
"Bash(wait:*)",
|
||||
"WebSearch",
|
||||
"Bash(SKILL_CREATOR_PATH=\"C:\\\\Users\\\\A00963355\\\\.claude\\\\plugins\\\\cache\\\\claude-plugins-official\\\\skill-creator\\\\d5c15b861cd2\\\\skills\\\\skill-creator\" && WORKSPACE=\"C:\\\\Users\\\\A00963355\\\\OneDrive - Amaris Zorggroep\\\\Documents\\\\VScodeProjects\\\\DragonsStash\\\\.claude\\\\skills\\\\tdlib-telegram-workspace\\\\iteration-1\" && python \"$SKILL_CREATOR_PATH/eval-viewer/generate_review.py\" \"$WORKSPACE\" --skill-name \"tdlib-telegram\" --benchmark \"$WORKSPACE/benchmark.json\" --static \"$WORKSPACE/review.html\" 2>&1)",
|
||||
"Bash(start:*)"
|
||||
"Bash(start:*)",
|
||||
"Bash(npm run:*)",
|
||||
"Bash(DATABASE_URL=\"postgresql://dragons:stash@localhost:5432/dragonsstash\" npx prisma migrate dev --name add-skipped-packages)",
|
||||
"Bash(git checkout:*)",
|
||||
"Bash(DATABASE_URL=\"postgresql://dragons:stash@localhost:5432/dragonsstash?schema=public\" npx prisma migrate dev --name add_package_groups 2>&1)",
|
||||
"Bash(psql:*)",
|
||||
"Bash(git log:*)",
|
||||
"Bash(git merge:*)"
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
+16
-1
@@ -54,9 +54,24 @@ steps:
|
||||
password:
|
||||
from_secret: gitea_password
|
||||
|
||||
- name: build-backup
|
||||
image: plugins/docker
|
||||
depends_on: [clone]
|
||||
settings:
|
||||
repo: git.samagsteribbe.nl/admin/dragonsstash-backup
|
||||
registry: git.samagsteribbe.nl
|
||||
dockerfile: backup/Dockerfile
|
||||
tags:
|
||||
- latest
|
||||
- "${DRONE_COMMIT_SHA:0:8}"
|
||||
username:
|
||||
from_secret: gitea_username
|
||||
password:
|
||||
from_secret: gitea_password
|
||||
|
||||
- name: deploy
|
||||
image: alpine
|
||||
depends_on: [build-app, build-worker, build-bot]
|
||||
depends_on: [build-app, build-worker, build-bot, build-backup]
|
||||
environment:
|
||||
SSH_KEY:
|
||||
from_secret: ssh_key
|
||||
|
||||
@@ -36,3 +36,12 @@ TDLIB_STATE_DIR="/data/tdlib"
|
||||
WORKER_MAX_ZIP_SIZE_MB=4096
|
||||
MULTIPART_TIMEOUT_HOURS=0
|
||||
LOG_LEVEL="info"
|
||||
|
||||
# Backup (NAS via SMB/CIFS + restic)
|
||||
NAS_HOST="" # Synology NAS IP or hostname reachable from this host
|
||||
NAS_SHARE="" # SMB share name, e.g. dragonsstash_backups
|
||||
NAS_USERNAME="" # SMB user with read/write on the share
|
||||
NAS_PASSWORD="" # SMB user password (avoid commas — they delimit cifs mount opts)
|
||||
RESTIC_PASSWORD="" # generate with: openssl rand -base64 32
|
||||
KUMA_PUSH_URL="" # optional: Uptime Kuma Push monitor URL; leave empty to disable alerting
|
||||
TZ="Etc/UTC"
|
||||
|
||||
@@ -55,3 +55,4 @@ src/generated
|
||||
nul
|
||||
tmpclaude-*
|
||||
.worktrees/
|
||||
worktrees/
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
# Disposable restore documentation fix report
|
||||
|
||||
## Scope
|
||||
|
||||
Updated `scripts/backup/README.md` only for the documentation change. This
|
||||
report is the requested verification artifact.
|
||||
|
||||
## Change
|
||||
|
||||
The monthly recovery rehearsal now documents a unique disposable Compose
|
||||
project with project-labeled volumes, snapshot selection, staging restore,
|
||||
PostgreSQL import, restoration of uploads and both TDLib volumes, disposable
|
||||
service startup, health and log checks, retained-file validation, known STL
|
||||
checksum and metadata comparison, narrowly scoped cleanup, and an evidence
|
||||
template.
|
||||
|
||||
The guide explicitly warns operators not to use production project names or
|
||||
volumes, and states that this documentation update did not run the rehearsal.
|
||||
|
||||
## Verification
|
||||
|
||||
- Focused text check: passed. Confirmed the guide contains the unique project
|
||||
warning, snapshot selection, staging restore, PostgreSQL import, all three
|
||||
protected volume restores, health check, retained-file validation, checksum
|
||||
comparison, scoped cleanup, evidence template, and the statement that the
|
||||
rehearsal was not run.
|
||||
- `git diff --check`: passed.
|
||||
- No recovery, Docker, Restic, PostgreSQL, or health-check commands were run;
|
||||
this change documents the operator procedure only.
|
||||
@@ -0,0 +1,37 @@
|
||||
# Backup Final Review Fix Report
|
||||
|
||||
## Implemented findings
|
||||
|
||||
- The backup container now validates that `RESTIC_REPOSITORY` resolves strictly
|
||||
below `/backup`, verifies that the Restic repository already has a readable
|
||||
configuration before a scheduled backup, and gives the explicit first-run
|
||||
initialization command when that preflight fails.
|
||||
- The backup manifest now records every successfully applied Prisma migration
|
||||
as a JSON object with its name and UTC completion timestamp. The `psql`
|
||||
query uses the existing `DATABASE_URL` connection configuration and stops on
|
||||
query errors.
|
||||
- Restore now requires `BACKUP_REPOSITORY`, validates that it resolves strictly
|
||||
below `/backup`, and validates the configured backup mount (including the
|
||||
existing writable probe) before `restore-live` can stop services or replace
|
||||
live data.
|
||||
- The backup runbook now documents explicit repository initialization, monthly
|
||||
full-read Restic checks, a disposable restore/checksum rehearsal, and
|
||||
post-restore Compose status/log checks.
|
||||
|
||||
## Verification
|
||||
|
||||
- `bash -n scripts/backup/container-entrypoint.sh`
|
||||
- `bash -n scripts/backup/restore.sh`
|
||||
- Focused `rg` assertions for repository validation/preflight, migration
|
||||
timestamp metadata, restore mount validation, initialization, maintenance,
|
||||
and post-restore runbook commands.
|
||||
- `npx prisma validate`
|
||||
- `git diff --check`
|
||||
|
||||
## Scope and concerns
|
||||
|
||||
- No Docker, NAS, systemd, Restic repository, or live restore was run, per the
|
||||
bounded review scope. The command-level behavior is therefore statically
|
||||
validated only.
|
||||
- Existing durable STL handling, the exact live-restore confirmation flag, and
|
||||
rollback/safety-artifact behavior were retained.
|
||||
@@ -0,0 +1,71 @@
|
||||
# Backup Scope-Correction Documentation Report
|
||||
|
||||
**Date:** 2026-07-22
|
||||
|
||||
## Changed files
|
||||
|
||||
- `docs/superpowers/specs/2026-07-21-database-and-file-backups-design.md`
|
||||
- `docs/superpowers/plans/2026-07-21-database-and-file-backups.md`
|
||||
- `.superpowers/sdd/scope-correction-docs-report.md`
|
||||
|
||||
## Rationale
|
||||
|
||||
The backup design and implementation plan now define the protected data set as a PostgreSQL logical dump plus the `tdlib_state` and `tdlib_bot_state` session volumes. They continue to require a host-restricted Synology NFS repository, Restic encryption, 30 daily snapshots, service quiescing, guarded restore, and session persistence.
|
||||
|
||||
`manual_uploads` and `tmp_zips` are explicitly excluded. The documents no longer require local retention of completed STL binaries, worker cleanup changes, database lifecycle fields for retained uploads, restored local STL files, or file-path validation. They state that STL binaries remain in Telegram and that the restored database preserves the metadata and mappings required to locate and send them.
|
||||
|
||||
Future Telegram channel-forwarding behavior and archive/STL-content integrity validation are explicitly identified as out of scope and future work.
|
||||
|
||||
## Checks run
|
||||
|
||||
- `git diff --check`
|
||||
- Scope scan of both documents for `manual_uploads`, `tmp_zips`, local STL retention, file-path validation, channel forwarding, and integrity language.
|
||||
- Reviewed the final diff to confirm the removed implementation work is limited to the specified backup-scope correction.
|
||||
- `git status --short` to confirm the commit stages only the two requested documents and this required report.
|
||||
|
||||
## Concerns
|
||||
|
||||
- This change intentionally updates documentation only. It does not modify backup scripts, Docker Compose, database schema, worker cleanup, or Telegram behavior.
|
||||
- A future implementation should validate its actual backup manifests and Compose mounts against this corrected plan before deployment.
|
||||
|
||||
---
|
||||
|
||||
# Monthly Backup Verification Documentation Follow-up
|
||||
|
||||
**Date:** 2026-07-22
|
||||
|
||||
## Fix
|
||||
|
||||
Aligned the approved backup design and implementation plan on the missing recurring operational work. The deployment operator now owns a documented monthly manual runbook task to run full `restic check --read-data`, perform a disposable restore rehearsal, and record the date, snapshot ID, integrity-check result, restore/health result, and cleanup result.
|
||||
|
||||
The plan verifies the first full-read check and rehearsal during acceptance, then carries the same procedure into the monthly runbook without adding a second systemd timer, script, or other production implementation. The Synology wording now precisely identifies the dedicated shared folder's NFS export as restricted to the Docker host's fixed IP.
|
||||
|
||||
## Scope preserved
|
||||
|
||||
The recovery set remains the PostgreSQL logical dump plus `tdlib_state` and `tdlib_bot_state` only. `manual_uploads`, STL binaries, archive/STL-content integrity, and future channel-forwarding checks remain outside this work.
|
||||
|
||||
## Checks completed
|
||||
|
||||
- `git diff --check` completed with no whitespace errors.
|
||||
- Focused assertions passed: `restic check --read-data` (4 matches), `deployment operator` (3), `disposable restore rehearsal` (6), `manual_uploads` (8), and `archive/STL-content integrity` (4) across the two approved documents.
|
||||
- Final diff review confirmed that this follow-up changes documentation only; no backup scripts, Compose configuration, or other production implementation files were modified.
|
||||
|
||||
---
|
||||
|
||||
# Monthly Backup Verification Review-Finding Fix
|
||||
|
||||
**Date:** 2026-07-22
|
||||
|
||||
## Fix
|
||||
|
||||
Confirmed and kept the approved design and implementation plan aligned on the reviewer finding: monthly recovery verification is an operator-owned operational task, consisting of a full `restic check --read-data` and a disposable restore rehearsal.
|
||||
|
||||
## Scope
|
||||
|
||||
The approved docs continue to limit the protected recovery set to the PostgreSQL logical dump, `tdlib_state`, and `tdlib_bot_state`. They do not add `manual_uploads`, STL-binary restore/checks, archive-content integrity checks, future channel-forwarding checks, or any new production timer/script.
|
||||
|
||||
## Checks
|
||||
|
||||
- `git diff --check`
|
||||
- Focused scope assertions over the two approved docs for monthly full Restic check/rehearsal wording and exclusions.
|
||||
- Staged-file review before commit to confirm the commit contains documentation/report files only.
|
||||
@@ -0,0 +1,240 @@
|
||||
# Scope correction implementation report
|
||||
|
||||
## Summary
|
||||
|
||||
Implemented the approved backup scope correction for Dragon's Stash. Backups and restores now cover only:
|
||||
|
||||
- PostgreSQL logical custom-format dump plus manifest/migration metadata.
|
||||
- `tdlib_state` worker Telegram session volume.
|
||||
- `tdlib_bot_state` bot Telegram session volume.
|
||||
|
||||
The implementation no longer treats `manual_uploads`, completed local STL binaries, or `tmp_zips` as protected backup data. Future channel forwarding and archive/STL-content integrity auditing remain out of scope.
|
||||
|
||||
## Changed files
|
||||
|
||||
- `docker-compose.yml`
|
||||
- Removed the backup service's read-only `manual_uploads:/data/uploads` mount.
|
||||
- Kept the normal operational app/worker `manual_uploads` mounts.
|
||||
- Kept both TDLib backup mounts.
|
||||
|
||||
- `scripts/backup/container-entrypoint.sh`
|
||||
- Removed `/data/uploads` as a required mounted directory.
|
||||
- Removed uploads from the manifest `volumePaths`.
|
||||
- Removed the `source:uploads` Restic tag.
|
||||
- Removed uploads from the Restic source list.
|
||||
- Kept database dump, manifest, worker TDLib, and bot TDLib sources/tags.
|
||||
|
||||
- `scripts/backup/restore.sh`
|
||||
- Removed restored uploads variables.
|
||||
- Removed uploads staging validation.
|
||||
- Removed manual uploads volume discovery, safety archive, replacement, and rollback.
|
||||
- Removed retained/manual upload file-path verification.
|
||||
- Removed temporary database verification that existed only for local upload file references.
|
||||
- Preserved guarded `restore-live` confirmation, backup mount/repository checks, service stop/start handling, safety PostgreSQL dump, TDLib volume safety archives, TDLib volume replacement/rollback, and `pg_restore --list`/`pg_restore --exit-on-error` validation.
|
||||
|
||||
- `prisma/schema.prisma`
|
||||
- Removed `ManualUploadFile.retainedAt`.
|
||||
|
||||
- `prisma/migrations/20260722100000_remove_retained_manual_files/migration.sql`
|
||||
- Added forward migration: `ALTER TABLE "manual_upload_files" DROP COLUMN IF EXISTS "retainedAt";`
|
||||
- Preserved the existing committed migration that added `retainedAt`.
|
||||
|
||||
- `src/app/api/uploads/route.ts`
|
||||
- Removed `retainedAt: new Date()` from manual upload file creation.
|
||||
|
||||
- `worker/src/manual-upload.ts`
|
||||
- Restored final best-effort cleanup of `/data/uploads/<uploadId>` using the older `path.join("/data/uploads", uploadId)` behavior.
|
||||
|
||||
- `scripts/backup/README.md`
|
||||
- Rewrote backup set and restore rehearsal docs around PostgreSQL plus both TDLib volumes only.
|
||||
- Removed local STL file, retainedAt, upload path, retained file reference, and restored checksum checks.
|
||||
- Clarified that STL binaries remain in Telegram and recovery preserves database mappings/Telegram IDs.
|
||||
- Kept monthly `restic check --read-data` and disposable restore rehearsal runbook.
|
||||
- Explicitly left future channel forwarding and archive/STL-content integrity auditing out of scope.
|
||||
|
||||
- `README.md`
|
||||
- Updated the production backup summary to name PostgreSQL logical dump plus Telegram session volumes as the protected set.
|
||||
- Clarified that `manual_uploads` and temporary ZIPs are excluded and STL binaries remain in Telegram.
|
||||
|
||||
## Verification
|
||||
|
||||
- `git diff --check`
|
||||
- Passed.
|
||||
|
||||
- `bash -n scripts/backup/container-entrypoint.sh scripts/backup/run-backup.sh scripts/backup/restore.sh`
|
||||
- Local `bash` failed because Windows only had the WSL shim and no installed WSL distribution.
|
||||
- Passed via Docker fallback:
|
||||
`docker run --rm --entrypoint bash -v E:\Projects\DragonsStash:/work:ro -w /work postgres:16-alpine -n scripts/backup/container-entrypoint.sh scripts/backup/run-backup.sh scripts/backup/restore.sh`
|
||||
|
||||
- `npx prisma validate`
|
||||
- Passed.
|
||||
|
||||
- `npm run build`
|
||||
- Passed.
|
||||
|
||||
- `cd worker && npm run build`
|
||||
- Passed.
|
||||
|
||||
- Focused backup/restore scope assertions
|
||||
- Backup shell paths assertion passed: no `manual_uploads`, `/data/uploads`, `retainedAt`, upload source tag, or upload-restore helper references in `scripts/backup/*.sh`.
|
||||
- Compose backup service assertion passed: no `manual_uploads`, `/data/uploads`, `retainedAt`, or `source:uploads` in the `backup` service block.
|
||||
- Active retainedAt assertion passed: no `retainedAt` in active Prisma schema, upload API, worker source, or backup shell scripts.
|
||||
- Active backup/restore upload-source assertion passed: no `manual_uploads`, `/data/uploads`, or `data/uploads` in backup/restore shell scripts.
|
||||
- Remaining expected matches are limited to normal operational app/worker upload mounts and paths, docs stating exclusions, and the historical add/drop migrations.
|
||||
|
||||
## Concerns
|
||||
|
||||
- None for implementation scope.
|
||||
- Environment note: local Bash is unavailable because WSL has no installed distribution; Bash syntax was verified inside Docker instead.
|
||||
|
||||
---
|
||||
|
||||
# Restore Path Scope Review-Finding Fix
|
||||
|
||||
**Date:** 2026-07-22
|
||||
|
||||
## Fix
|
||||
|
||||
Addressed the Important restore finding by replacing both unrestricted
|
||||
`restic restore` calls in `scripts/backup/restore.sh` with a shared filtered
|
||||
restore wrapper. The wrapper restores only the backup source paths actually
|
||||
written by `scripts/backup/container-entrypoint.sh`:
|
||||
|
||||
- `/staging/backup-*/database.dump`
|
||||
- `/staging/backup-*/manifest` and `/staging/backup-*/manifest/**`
|
||||
- `/data/tdlib-worker` and `/data/tdlib-worker/**`
|
||||
- `/data/tdlib-bot` and `/data/tdlib-bot/**`
|
||||
|
||||
Added an explicit restored-tree guard that refuses unexpected restored content:
|
||||
top-level restored directories other than `staging` and `data`, direct
|
||||
`data/*` entries other than `tdlib-worker` and `tdlib-bot`, and direct
|
||||
`staging/backup-*/*` entries other than `database.dump` and `manifest`. This
|
||||
rejects old/broad snapshots that would otherwise restore `data/uploads`,
|
||||
temporary ZIP or database volume trees, or other unexpected volume content.
|
||||
|
||||
Preserved the existing guarded live-restore confirmation, staging-directory
|
||||
validation, mount/repository checks, snapshot verification, custom
|
||||
PostgreSQL-dump validation, service stop/start lifecycle, health check, safety
|
||||
database dump, TDLib safety archives, and rollback of exactly the PostgreSQL
|
||||
database plus the two TDLib volumes.
|
||||
|
||||
Added `scripts/backup/restore-path-assertions.sh`, a focused shell assertion
|
||||
harness that stubs Docker/Restic and verifies both staging and live restore use
|
||||
the expected include filters and that unexpected restored data-volume content is
|
||||
rejected explicitly.
|
||||
|
||||
Addressed the Minor documentation gap in the root README backup section by
|
||||
stating that forwarding behavior and archive/STL-content integrity auditing are
|
||||
future work outside the backup scope.
|
||||
|
||||
## Verification
|
||||
|
||||
- Red check before implementation:
|
||||
|
||||
```text
|
||||
& 'C:\Program Files\Git\bin\bash.exe' -lc 'scripts/backup/restore-path-assertions.sh'
|
||||
ASSERTION FAILED: database dump include filter missing
|
||||
```
|
||||
|
||||
- Bash syntax check:
|
||||
|
||||
```text
|
||||
& 'C:\Program Files\Git\bin\bash.exe' -lc 'bash -n scripts/backup/container-entrypoint.sh scripts/backup/run-backup.sh scripts/backup/restore.sh scripts/backup/restore-path-assertions.sh'
|
||||
[passed with no output]
|
||||
```
|
||||
|
||||
- Whitespace check:
|
||||
|
||||
```text
|
||||
git diff --check
|
||||
warning: in the working copy of '.superpowers/sdd/scope-correction-implementation-report.md', LF will be replaced by CRLF the next time Git touches it
|
||||
warning: in the working copy of 'README.md', LF will be replaced by CRLF the next time Git touches it
|
||||
warning: in the working copy of 'scripts/backup/restore.sh', LF will be replaced by CRLF the next time Git touches it
|
||||
[exit 0]
|
||||
```
|
||||
|
||||
- Focused restore-path assertions:
|
||||
|
||||
```text
|
||||
& 'C:\Program Files\Git\bin\bash.exe' -lc 'scripts/backup/restore-path-assertions.sh'
|
||||
restore-path assertions passed
|
||||
```
|
||||
|
||||
## Concerns
|
||||
|
||||
- None for implementation scope.
|
||||
- Git Bash was available and used for shell syntax/assertion checks, so Docker
|
||||
fallback was not needed for the final syntax verification.
|
||||
|
||||
---
|
||||
|
||||
# Important Operational Findings Fix
|
||||
|
||||
**Date:** 2026-07-22
|
||||
|
||||
## Fix
|
||||
|
||||
Addressed the two Important operational findings from final review:
|
||||
|
||||
- `scripts/backup/run-backup.sh`
|
||||
- Backup wrapper restart failures now make an otherwise successful backup
|
||||
exit non-zero.
|
||||
- Existing non-zero backup failures remain preserved if service restart also
|
||||
fails.
|
||||
- Added `scripts/backup/run-backup-assertions.sh` to assert both exit-code
|
||||
cases with a fake Docker/Compose environment.
|
||||
|
||||
- `scripts/backup/restore.sh`
|
||||
- `restore-live` now captures the managed services that were running before
|
||||
live restore using `docker compose --profile full ps --status running`.
|
||||
- Live restore stops only those previously running managed services.
|
||||
- Successful live restore starts only those previously running services, so
|
||||
the profile-gated optional `bot` is not started if it was not running.
|
||||
- Failure handling leaves services stopped and still rolls back only the
|
||||
PostgreSQL database plus both TDLib volumes.
|
||||
- App health wait now runs only when `app` was previously running.
|
||||
- Extended `scripts/backup/restore-path-assertions.sh` to assert subset
|
||||
stop/start behavior and skipped health checks when `app` was not running.
|
||||
|
||||
Addressed the Minor staging-path documentation mismatch by aligning
|
||||
`.env.example` with the backup README example:
|
||||
`/var/lib/dragons-stash/backup-staging`.
|
||||
|
||||
## Verification
|
||||
|
||||
- Red checks before implementation:
|
||||
- `docker run --rm -v "${PWD}:/work" -w /work ubuntu:24.04 bash scripts/backup/run-backup-assertions.sh`
|
||||
- Failed as expected: backup reported success when restart failed.
|
||||
- `docker run --rm -v "${PWD}:/work" -w /work ubuntu:24.04 bash scripts/backup/restore-path-assertions.sh`
|
||||
- Failed as expected: restore-live stopped the fixed `app worker bot`
|
||||
service set instead of the running subset.
|
||||
|
||||
- Focused assertions after implementation:
|
||||
- `docker run --rm -v "${PWD}:/work" -w /work ubuntu:24.04 bash scripts/backup/run-backup-assertions.sh`
|
||||
- Passed.
|
||||
- `docker run --rm -v "${PWD}:/work" -w /work ubuntu:24.04 bash scripts/backup/restore-path-assertions.sh`
|
||||
- Passed.
|
||||
|
||||
- `git diff --check`
|
||||
- Passed.
|
||||
|
||||
- Docker Bash syntax check:
|
||||
- `docker run --rm -v "${PWD}:/work" -w /work ubuntu:24.04 bash -n scripts/backup/run-backup.sh scripts/backup/restore.sh scripts/backup/container-entrypoint.sh scripts/backup/restore-path-assertions.sh scripts/backup/run-backup-assertions.sh`
|
||||
- Passed.
|
||||
|
||||
- `npx prisma validate`
|
||||
- Passed.
|
||||
|
||||
- `npm run build`
|
||||
- Passed.
|
||||
|
||||
- `npm run lint`
|
||||
- Failed on unrelated existing React lint issues in `src/` and mirrored
|
||||
`.worktrees/worker-improvements` files; no failures were in touched backup
|
||||
files.
|
||||
|
||||
## Concerns
|
||||
|
||||
- Local `bash` is unavailable because the Windows `bash` command resolves to a
|
||||
WSL shim with no installed distribution; shell checks used Docker fallback.
|
||||
- Full `npm run lint` remains blocked by pre-existing unrelated lint errors.
|
||||
@@ -140,6 +140,18 @@ docker compose --profile bot up -d
|
||||
> **Tip:** Create a bot token via [@BotFather](https://t.me/BotFather) on Telegram and set `BOT_TOKEN` in `.env`.
|
||||
> Get Telegram API credentials from [my.telegram.org/apps](https://my.telegram.org/apps).
|
||||
|
||||
### Production Backups
|
||||
|
||||
Docker volumes are not backups. Production backups protect a PostgreSQL
|
||||
logical dump plus the worker and bot Telegram session volumes in an encrypted
|
||||
Restic repository on a Synology NFS share. `manual_uploads` and temporary ZIP
|
||||
processing data are excluded; STL binaries remain in Telegram, while the
|
||||
database mappings and Telegram IDs are what recovery preserves for lookup and
|
||||
delivery. Forwarding behavior and archive/STL-content integrity auditing are
|
||||
future work outside this backup scope. See the [backup and recovery guide](scripts/backup/README.md)
|
||||
for Synology setup, secrets, systemd installation, monitoring, retention, and
|
||||
guarded restore procedures.
|
||||
|
||||
### Seeding the Database
|
||||
|
||||
To seed the database with sample data on first run:
|
||||
|
||||
@@ -0,0 +1,13 @@
|
||||
FROM alpine:3.20
|
||||
|
||||
# Note: use busybox's built-in crond (Alpine base), NOT the dcron package —
|
||||
# dcron's crond fails with "setpgid: Operation not permitted" in this runtime.
|
||||
RUN apk add --no-cache restic postgresql16-client curl tzdata tar bash
|
||||
|
||||
COPY backup/backup.sh /backup.sh
|
||||
COPY backup/entrypoint.sh /entrypoint.sh
|
||||
COPY backup/crontab /etc/crontabs/root
|
||||
|
||||
RUN chmod +x /backup.sh /entrypoint.sh
|
||||
|
||||
ENTRYPOINT ["/entrypoint.sh"]
|
||||
@@ -0,0 +1,32 @@
|
||||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
|
||||
report_failure() {
|
||||
[ -n "${KUMA_PUSH_URL:-}" ] || return 0
|
||||
curl -fsS "$KUMA_PUSH_URL" --get \
|
||||
--data-urlencode "status=down" \
|
||||
--data-urlencode "msg=$BASH_COMMAND failed" || true
|
||||
}
|
||||
trap report_failure ERR
|
||||
|
||||
DUMP_FILE=/tmp/dragonsstash.dump
|
||||
TAR_FILE=/tmp/tdlib.tar.gz
|
||||
|
||||
trap 'rm -f "$DUMP_FILE" "$TAR_FILE"' EXIT
|
||||
|
||||
pg_dump -h dragonsstash-db -U "$POSTGRES_USER" -d "$POSTGRES_DB" -Fc -f "$DUMP_FILE"
|
||||
|
||||
# TDLib volumes are tarred live (best-effort, per design). A file changing
|
||||
# mid-read makes GNU tar exit 1 (warning) — that is expected here and must not
|
||||
# abort the backup. Only a genuine error (exit >= 2) is fatal.
|
||||
tar --warning=no-file-changed -czf "$TAR_FILE" -C /data tdlib-worker tdlib-bot \
|
||||
|| { rc=$?; [ "$rc" -le 1 ] || exit "$rc"; }
|
||||
|
||||
restic backup "$DUMP_FILE" "$TAR_FILE"
|
||||
restic forget --keep-daily 14 --prune
|
||||
|
||||
if [ -n "${KUMA_PUSH_URL:-}" ]; then
|
||||
curl -fsS "$KUMA_PUSH_URL" --get \
|
||||
--data-urlencode "status=up" \
|
||||
--data-urlencode "msg=OK"
|
||||
fi
|
||||
@@ -0,0 +1,2 @@
|
||||
0 3 * * * /backup.sh >> /proc/1/fd/1 2>&1
|
||||
0 4 * * 0 restic check >> /proc/1/fd/1 2>&1
|
||||
@@ -0,0 +1,11 @@
|
||||
#!/bin/bash
|
||||
set -uo pipefail
|
||||
|
||||
# Ensure the repo exists, but never crash-loop on it: a transient error reading
|
||||
# the repo (CIFS hiccup, stale lock) must not kill PID 1. `restic init` failing
|
||||
# because the repo already exists is expected and harmless here.
|
||||
if ! restic cat config >/dev/null 2>&1; then
|
||||
restic init || echo "restic init skipped (repo already exists or temporarily unreachable)"
|
||||
fi
|
||||
|
||||
exec crond -f -l 2
|
||||
@@ -0,0 +1,13 @@
|
||||
[Unit]
|
||||
Description=Dragon's Stash off-host backup
|
||||
After=network-online.target docker.service
|
||||
Requires=docker.service
|
||||
Wants=network-online.target
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
User=root
|
||||
EnvironmentFile=-/etc/dragons-stash/backup.env
|
||||
WorkingDirectory=/opt/stacks/DragonsStash
|
||||
ExecStart=/opt/stacks/DragonsStash/scripts/backup/run-backup.sh
|
||||
TimeoutStartSec=infinity
|
||||
@@ -0,0 +1,11 @@
|
||||
[Unit]
|
||||
Description=Nightly Dragon's Stash off-host backup
|
||||
|
||||
[Timer]
|
||||
OnCalendar=*-*-* 03:00:00
|
||||
Persistent=true
|
||||
RandomizedDelaySec=15m
|
||||
Unit=dragons-stash-backup.service
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
+38
-2
@@ -97,6 +97,35 @@ services:
|
||||
networks:
|
||||
- backend
|
||||
|
||||
backup:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: backup/Dockerfile
|
||||
pull_policy: never
|
||||
environment:
|
||||
- POSTGRES_USER=${POSTGRES_USER:-dragons}
|
||||
- POSTGRES_PASSWORD=${POSTGRES_PASSWORD:-stash}
|
||||
- PGPASSWORD=${POSTGRES_PASSWORD:-stash}
|
||||
- POSTGRES_DB=${POSTGRES_DB:-dragonsstash}
|
||||
- RESTIC_REPOSITORY=/backups/restic-repo
|
||||
- RESTIC_PASSWORD=${RESTIC_PASSWORD:?Set RESTIC_PASSWORD in .env}
|
||||
- KUMA_PUSH_URL=${KUMA_PUSH_URL:-}
|
||||
- TZ=${TZ:-Etc/UTC}
|
||||
volumes:
|
||||
- tdlib_state:/data/tdlib-worker:ro
|
||||
- tdlib_bot_state:/data/tdlib-bot:ro
|
||||
- nas_backups:/backups
|
||||
depends_on:
|
||||
db:
|
||||
condition: service_healthy
|
||||
restart: unless-stopped
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 1G
|
||||
networks:
|
||||
- backend
|
||||
|
||||
db:
|
||||
image: postgres:16-alpine
|
||||
environment:
|
||||
@@ -116,8 +145,10 @@ services:
|
||||
limits:
|
||||
memory: 1G
|
||||
networks:
|
||||
- frontend
|
||||
- backend
|
||||
frontend: {}
|
||||
backend:
|
||||
aliases:
|
||||
- dragonsstash-db
|
||||
|
||||
volumes:
|
||||
postgres_data:
|
||||
@@ -125,6 +156,11 @@ volumes:
|
||||
tdlib_bot_state:
|
||||
tmp_zips:
|
||||
manual_uploads:
|
||||
nas_backups:
|
||||
driver_opts:
|
||||
type: cifs
|
||||
o: "username=${NAS_USERNAME},password=${NAS_PASSWORD},vers=3.0,uid=0,gid=0,file_mode=0660,dir_mode=0770"
|
||||
device: "//${NAS_HOST}/${NAS_SHARE}"
|
||||
|
||||
networks:
|
||||
frontend:
|
||||
|
||||
@@ -0,0 +1,450 @@
|
||||
# Database and File Backups Implementation Plan
|
||||
|
||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
||||
|
||||
**Goal:** Add a nightly, encrypted disaster-recovery backup for PostgreSQL and Telegram session volumes, stored on a Synology NAS.
|
||||
|
||||
**Architecture:** A Linux-host systemd timer invokes a host orchestration script. The script verifies the mounted Synology NFS share, stops the app/worker/bot services for consistency, and runs a one-shot Docker Compose backup service. The backup service creates a PostgreSQL custom-format dump and stores it with the two TDLib session volumes in a Restic repository on the NAS. A guarded restore command reconstructs the database and sessions, and documentation describes setup and testing. STL binaries remain in Telegram; restored database metadata and mappings continue to identify the Telegram content used for lookup and delivery.
|
||||
|
||||
**Tech Stack:** Docker Compose, PostgreSQL 16 `pg_dump`/`pg_restore`, Restic repository encryption and retention, Synology NFS, Linux systemd service/timer, Bash.
|
||||
|
||||
## Global Constraints
|
||||
|
||||
- PostgreSQL data must be backed up as a logical custom-format dump; the raw `postgres_data` volume is not the primary backup.
|
||||
- The `tdlib_state` and `tdlib_bot_state` volumes are included in every successful snapshot.
|
||||
- The `manual_uploads` and `tmp_zips` volumes are excluded.
|
||||
- Completed STL binaries are not retained locally for backup; existing worker cleanup behavior remains unchanged. Telegram remains the binary store, while PostgreSQL retains the metadata and mappings needed to locate and send the files after restore.
|
||||
- The Docker host must stop `app`, `worker`, and `bot` while session volumes are captured; PostgreSQL remains running for `pg_dump`.
|
||||
- The backup repository is encrypted and stored on a Synology NFS share restricted to the Docker host.
|
||||
- Retention is 30 daily snapshots; pruning is allowed only after a verified successful backup.
|
||||
- A failed run must restart services and preserve the last known-good snapshot.
|
||||
- A guarded restore must require explicit confirmation before replacing live database or session-volume data.
|
||||
- No in-app backup UI is part of this implementation.
|
||||
- Future Telegram channel-forwarding behavior and archive/STL-content integrity validation are explicitly out of scope.
|
||||
- The repository has no automated test framework; verification uses Bash syntax checks, Docker Compose validation, logs, Restic checks, and a disposable restore rehearsal.
|
||||
|
||||
---
|
||||
|
||||
## File and Responsibility Map
|
||||
|
||||
Create or modify only these focused units:
|
||||
|
||||
- Create `backup/Dockerfile`: build the one-shot image containing PostgreSQL client tools, Restic, Bash, and the backup entrypoint.
|
||||
- Create `scripts/backup/container-entrypoint.sh`: run the backup inside Compose, including dump creation, manifest creation, Restic snapshot, verification, and retention.
|
||||
- Create `scripts/backup/run-backup.sh`: host-level lock, NFS mount validation, service stop/start, and invocation of the one-shot Compose service.
|
||||
- Create `scripts/backup/restore.sh`: guarded restore orchestration for a selected Restic snapshot.
|
||||
- Create `deploy/systemd/dragons-stash-backup.service`: systemd unit invoking the host backup script.
|
||||
- Create `deploy/systemd/dragons-stash-backup.timer`: nightly schedule.
|
||||
- Create `scripts/backup/README.md`: Synology setup, host mount, secrets, first backup, restore, and operational troubleshooting.
|
||||
- Modify `docker-compose.yml`: add the profile-gated one-shot `backup` service and its read-only session-volume mounts.
|
||||
- Modify `.env.example`: document backup mount, staging, repository, and secret-file configuration without committing secrets.
|
||||
- Modify `README.md`: add the production backup setup and restore entry points, linking to the detailed backup guide.
|
||||
|
||||
---
|
||||
|
||||
### Task 1: Add the backup service and configuration contract
|
||||
|
||||
**Files:**
|
||||
- Modify: `docker-compose.yml`
|
||||
- Modify: `.env.example`
|
||||
- Create: `backup/Dockerfile`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: existing `db`, `tdlib_state`, and `tdlib_bot_state` Compose resources.
|
||||
- Produces: a profile-gated Compose service named `backup` that mounts the two session volumes read-only, connects to the `backend` network, and exposes `/backup` and `/staging` to the container entrypoint.
|
||||
|
||||
- [ ] **Step 1: Add explicit backup environment variables to `.env.example`**
|
||||
|
||||
Add this block without real credentials:
|
||||
|
||||
```dotenv
|
||||
# Disaster recovery backups
|
||||
BACKUP_MOUNT_PATH="/mnt/dragonsstash-backups"
|
||||
BACKUP_STAGING_PATH="/var/lib/dragons-stash-backup/staging"
|
||||
BACKUP_REPOSITORY="/backup/restic"
|
||||
BACKUP_RESTIC_PASSWORD_FILE="/etc/dragons-stash/restic-password"
|
||||
BACKUP_RETENTION_DAYS=30
|
||||
BACKUP_APP_VERSION="unknown"
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Add the profile-gated `backup` service to `docker-compose.yml`**
|
||||
|
||||
Add a service with these properties:
|
||||
|
||||
```yaml
|
||||
backup:
|
||||
profiles: ["backup"]
|
||||
build:
|
||||
context: .
|
||||
dockerfile: backup/Dockerfile
|
||||
environment:
|
||||
DATABASE_URL: postgresql://${POSTGRES_USER:-dragons}:${POSTGRES_PASSWORD:-stash}@db:5432/${POSTGRES_DB:-dragonsstash}
|
||||
RESTIC_REPOSITORY: ${BACKUP_REPOSITORY:-/backup/restic}
|
||||
RESTIC_PASSWORD_FILE: /run/secrets/restic-password
|
||||
BACKUP_RETENTION_DAYS: ${BACKUP_RETENTION_DAYS:-30}
|
||||
BACKUP_APP_VERSION: ${BACKUP_APP_VERSION:-unknown}
|
||||
user: "0:0"
|
||||
volumes:
|
||||
- tdlib_state:/data/tdlib-worker:ro
|
||||
- tdlib_bot_state:/data/tdlib-bot:ro
|
||||
- ${BACKUP_MOUNT_PATH:?Set BACKUP_MOUNT_PATH to the mounted Synology share}:/backup:rw
|
||||
- ${BACKUP_STAGING_PATH:?Set BACKUP_STAGING_PATH to a local staging directory}:/staging:rw
|
||||
- ${BACKUP_RESTIC_PASSWORD_FILE:?Set BACKUP_RESTIC_PASSWORD_FILE to a root-readable secret file}:/run/secrets/restic-password:ro
|
||||
depends_on:
|
||||
db:
|
||||
condition: service_healthy
|
||||
networks:
|
||||
- backend
|
||||
```
|
||||
|
||||
Do not mount `manual_uploads` or `tmp_zips`. Ensure the new service does not have `restart: always` and is not started by the normal production `docker compose up -d` command unless the `backup` profile is explicitly requested.
|
||||
|
||||
- [ ] **Step 3: Create the backup image definition**
|
||||
|
||||
Create `backup/Dockerfile`:
|
||||
|
||||
```dockerfile
|
||||
FROM postgres:16-alpine
|
||||
|
||||
RUN apk add --no-cache bash restic coreutils
|
||||
|
||||
COPY scripts/backup/container-entrypoint.sh /usr/local/bin/dragons-stash-backup
|
||||
RUN chmod 0755 /usr/local/bin/dragons-stash-backup
|
||||
|
||||
ENTRYPOINT ["/usr/local/bin/dragons-stash-backup"]
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Validate the Compose contract**
|
||||
|
||||
Run on a Linux host with the required variables available:
|
||||
|
||||
```bash
|
||||
docker compose --profile backup config --quiet
|
||||
```
|
||||
|
||||
Expected: exit code `0` and no Compose validation errors. If the required NAS/secret paths are absent, the command must fail with the explicit variable-name error rather than silently using a host path.
|
||||
|
||||
- [ ] **Step 5: Commit the service boundary**
|
||||
|
||||
```bash
|
||||
git add backup/Dockerfile docker-compose.yml .env.example
|
||||
git commit -m "feat: add backup compose service"
|
||||
```
|
||||
|
||||
### Task 2: Implement the one-shot backup container
|
||||
|
||||
**Files:**
|
||||
- Create: `scripts/backup/container-entrypoint.sh`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `DATABASE_URL`, `RESTIC_REPOSITORY`, `RESTIC_PASSWORD_FILE`, `BACKUP_RETENTION_DAYS`, `/data/tdlib-worker`, `/data/tdlib-bot`, `/backup`, and `/staging`.
|
||||
- Produces: exit `0` only after a verified Restic snapshot and successful retention pruning; non-zero on any failed dump, snapshot, verification, or prune step.
|
||||
|
||||
- [ ] **Step 1: Define strict shell behavior and required inputs**
|
||||
|
||||
The script must begin with:
|
||||
|
||||
```bash
|
||||
#!/usr/bin/env bash
|
||||
set -Eeuo pipefail
|
||||
```
|
||||
|
||||
Validate that `DATABASE_URL`, `RESTIC_REPOSITORY`, `RESTIC_PASSWORD_FILE`, and `BACKUP_RETENTION_DAYS` are set, that the password file is readable, and that `/backup` and `/staging` are mounted directories.
|
||||
|
||||
- [ ] **Step 2: Create a per-run staging directory and cleanup trap**
|
||||
|
||||
Use a directory below `/staging` named with UTC timestamp and process ID. Register an `EXIT` trap that removes only that directory. Never remove `/staging` itself or any directory under `/backup`.
|
||||
|
||||
- [ ] **Step 3: Create the PostgreSQL dump**
|
||||
|
||||
Run `pg_dump` using the connection URL and custom format:
|
||||
|
||||
```bash
|
||||
pg_dump --format=custom --file="$RUN_DIR/database.dump" "$DATABASE_URL"
|
||||
```
|
||||
|
||||
After the command succeeds, require the dump to be a non-empty regular file. Generate a SHA-256 checksum for the dump in the manifest directory.
|
||||
|
||||
- [ ] **Step 4: Create the manifest**
|
||||
|
||||
Write a JSON manifest containing the UTC backup timestamp, repository path, retention value, dump filename, dump checksum, and the two TDLib volume paths captured. Obtain the application image/version from an explicit `BACKUP_APP_VERSION` environment value when supplied; otherwise record `unknown` rather than guessing from mutable container state.
|
||||
|
||||
- [ ] **Step 5: Create one Restic snapshot**
|
||||
|
||||
Run one `restic backup` command against the staged database dump, manifest, and the two mounted persistent session volumes. Use stable source labels so the snapshot can be recognized during restore. Do not mount or include `manual_uploads` or `tmp_zips`.
|
||||
|
||||
- [ ] **Step 6: Verify and apply retention**
|
||||
|
||||
After `restic backup` succeeds:
|
||||
|
||||
```bash
|
||||
restic snapshots --latest 1
|
||||
restic check
|
||||
restic forget --keep-daily "$BACKUP_RETENTION_DAYS" --prune
|
||||
```
|
||||
|
||||
If any command fails, exit non-zero and do not run `forget --prune`. The host wrapper will restart the stopped services. When invoked with an unrecognized first argument, the entrypoint must pass the remaining arguments to the `restic` binary so operators can inspect the repository through the Compose image without installing Restic on the host:
|
||||
|
||||
```bash
|
||||
case "${1:-backup}" in
|
||||
backup) run_backup ;;
|
||||
restore) run_restore "$@" ;;
|
||||
*) exec restic "$@" ;;
|
||||
esac
|
||||
```
|
||||
|
||||
- [ ] **Step 7: Build and run a container-only smoke test**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
docker build -f backup/Dockerfile -t dragons-stash-backup:smoke .
|
||||
bash -n scripts/backup/container-entrypoint.sh
|
||||
```
|
||||
|
||||
Expected: image build succeeds and Bash reports no syntax errors. The full snapshot test waits until the host wrapper and a real PostgreSQL/session-volume environment exist.
|
||||
|
||||
- [ ] **Step 8: Commit the backup container**
|
||||
|
||||
```bash
|
||||
git add backup/Dockerfile scripts/backup/container-entrypoint.sh
|
||||
git commit -m "feat: implement encrypted database and session snapshots"
|
||||
```
|
||||
|
||||
### Task 3: Add the host orchestration script and nightly systemd timer
|
||||
|
||||
**Files:**
|
||||
- Create: `scripts/backup/run-backup.sh`
|
||||
- Create: `deploy/systemd/dragons-stash-backup.service`
|
||||
- Create: `deploy/systemd/dragons-stash-backup.timer`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `.env`/deployment environment, the mounted `BACKUP_MOUNT_PATH`, Docker Compose project, and the `backup` service from Task 1.
|
||||
- Produces: one host command that safely stops and restarts services and returns the backup container's exit status; systemd runs it nightly.
|
||||
|
||||
- [ ] **Step 1: Implement lock and mount validation**
|
||||
|
||||
The host script must use `flock` on `/run/lock/dragons-stash-backup.lock`, reject a concurrent run, and validate the NAS mount with both `mountpoint --q "$BACKUP_MOUNT_PATH"` and a writable probe file that is immediately removed. A local directory at the same path must not pass validation. The script reads `BACKUP_MOUNT_PATH`, `BACKUP_STAGING_PATH`, `BACKUP_RESTIC_PASSWORD_FILE`, and `BACKUP_RETENTION_DAYS` from the systemd environment file.
|
||||
|
||||
- [ ] **Step 2: Capture service state and define guaranteed restart**
|
||||
|
||||
Before stopping services, record which of `app`, `worker`, and `bot` are running with `docker compose ps --status running -q SERVICE`. Stop only the services that were running. Register an `EXIT` trap that starts exactly those services and preserves the backup command's original exit code.
|
||||
|
||||
- [ ] **Step 3: Invoke the profile-gated backup service**
|
||||
|
||||
After the services stop, run:
|
||||
|
||||
```bash
|
||||
docker compose --profile backup run --rm backup backup
|
||||
```
|
||||
|
||||
Pass through the container exit code. The wrapper must not call the Restic retention command itself; that responsibility stays inside the backup container.
|
||||
|
||||
- [ ] **Step 4: Add the systemd service**
|
||||
|
||||
Create a unit with `Type=oneshot`, `User=root`, `EnvironmentFile=-/etc/dragons-stash/backup.env`, `WorkingDirectory` set to the production Compose directory, `ExecStart` pointing to the absolute `run-backup.sh` path, and `TimeoutStartSec=infinity`. Configure `After=network-online.target docker.service` and `Requires=docker.service`. Do not put the Restic password or database password in the unit file.
|
||||
|
||||
- [ ] **Step 5: Add the nightly timer**
|
||||
|
||||
Create a timer using `OnCalendar=*-*-* 03:00:00`, `Persistent=true`, and `RandomizedDelaySec=15m`. Set `Unit=dragons-stash-backup.service` and `WantedBy=timers.target`.
|
||||
|
||||
- [ ] **Step 6: Validate shell and systemd files**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
bash -n scripts/backup/run-backup.sh
|
||||
systemd-analyze verify deploy/systemd/dragons-stash-backup.service deploy/systemd/dragons-stash-backup.timer
|
||||
```
|
||||
|
||||
Expected: both commands exit `0`. Run `systemctl list-timers dragons-stash-backup.timer` after installation and confirm the next run is scheduled.
|
||||
|
||||
- [ ] **Step 7: Commit scheduling and orchestration**
|
||||
|
||||
```bash
|
||||
git add scripts/backup/run-backup.sh deploy/systemd/dragons-stash-backup.service deploy/systemd/dragons-stash-backup.timer
|
||||
git commit -m "feat: schedule nightly off-host backups"
|
||||
```
|
||||
|
||||
### Task 4: Implement guarded restore tooling
|
||||
|
||||
**Files:**
|
||||
- Create: `scripts/backup/restore.sh`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: a Restic snapshot ID, the same repository/password configuration, the backup Compose service, and the live Compose project.
|
||||
- Produces: restored PostgreSQL data and Telegram session volumes only after explicit confirmation for live replacement; a non-destructive staging restore by default.
|
||||
|
||||
- [ ] **Step 1: Define restore modes and destructive guard**
|
||||
|
||||
Support these commands:
|
||||
|
||||
```bash
|
||||
./scripts/backup/restore.sh list
|
||||
./scripts/backup/restore.sh verify SNAPSHOT_ID
|
||||
./scripts/backup/restore.sh restore-to-staging SNAPSHOT_ID STAGING_DIR
|
||||
./scripts/backup/restore.sh restore-live SNAPSHOT_ID --confirm-replace-live-data
|
||||
```
|
||||
|
||||
Reject `restore-live` unless the exact confirmation flag is present. `list`, `verify`, and `restore-to-staging` must not stop services or modify live volumes.
|
||||
|
||||
- [ ] **Step 2: Implement snapshot verification and staging restore**
|
||||
|
||||
Use `restic snapshots`, `restic check`, and `restic restore SNAPSHOT_ID --target STAGING_DIR`. Verify that the restored staging tree contains a non-empty custom-format dump, a manifest, `tdlib-worker`, and `tdlib-bot` before reporting success. Do not add file-path checks, binary checksums, archive/STL-content validation, or channel-forwarding behavior.
|
||||
|
||||
- [ ] **Step 3: Implement live restore sequencing**
|
||||
|
||||
For `restore-live`:
|
||||
|
||||
1. Confirm the Compose project and target repository.
|
||||
2. Stop `app`, `worker`, and `bot`.
|
||||
3. Create a safety PostgreSQL dump of the current database into local staging.
|
||||
4. Restore the selected snapshot to a separate staging directory.
|
||||
5. Replace the two Docker session volumes only after the restored tree passes validation.
|
||||
6. Recreate the configured database from the restored custom-format dump using `pg_restore --no-owner`.
|
||||
7. Start services and run the health endpoint plus worker/bot startup and authentication checks.
|
||||
|
||||
If any step fails, leave the services stopped, print the exact staging path and failure, and do not delete the safety dump.
|
||||
|
||||
- [ ] **Step 4: Validate the restore command without touching live data**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
bash -n scripts/backup/restore.sh
|
||||
./scripts/backup/restore.sh list
|
||||
```
|
||||
|
||||
Expected: syntax passes and `list` prints available snapshot IDs without stopping any service or modifying a volume.
|
||||
|
||||
- [ ] **Step 5: Commit guarded restore tooling**
|
||||
|
||||
```bash
|
||||
git add scripts/backup/restore.sh
|
||||
git commit -m "feat: add guarded database and session restore"
|
||||
```
|
||||
|
||||
### Task 5: Document Synology setup, operations, and recovery
|
||||
|
||||
**Files:**
|
||||
- Create: `scripts/backup/README.md`
|
||||
- Modify: `README.md`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: the exact environment variables, systemd units, and restore commands from Tasks 1-4.
|
||||
- Produces: operator-facing instructions that do not require reading implementation files.
|
||||
|
||||
- [ ] **Step 1: Document Synology configuration**
|
||||
|
||||
Document creating the `dragonsstash-backups` shared folder, enabling NFS, and configuring that shared folder's NFS export to allow only the Docker host's fixed IP. Document mounting it at `/mnt/dragonsstash-backups`. Include commands for checking the mount:
|
||||
|
||||
```bash
|
||||
mountpoint /mnt/dragonsstash-backups
|
||||
touch /mnt/dragonsstash-backups/.write-test
|
||||
rm /mnt/dragonsstash-backups/.write-test
|
||||
```
|
||||
|
||||
Do not document exposing NFS to the Internet.
|
||||
|
||||
- [ ] **Step 2: Document secret and staging setup**
|
||||
|
||||
Document creating the root-readable Restic password file at `/etc/dragons-stash/restic-password`, creating the local staging directory, setting ownership/permissions, and adding the backup variables to the production environment without committing secrets.
|
||||
|
||||
- [ ] **Step 3: Document installation and first-run commands**
|
||||
|
||||
Include:
|
||||
|
||||
```bash
|
||||
sudo install -m 0644 deploy/systemd/dragons-stash-backup.service /etc/systemd/system/
|
||||
sudo install -m 0644 deploy/systemd/dragons-stash-backup.timer /etc/systemd/system/
|
||||
sudo systemctl daemon-reload
|
||||
sudo systemctl enable --now dragons-stash-backup.timer
|
||||
sudo systemctl start dragons-stash-backup.service
|
||||
sudo journalctl -u dragons-stash-backup.service -n 100 --no-pager
|
||||
```
|
||||
|
||||
Explain that the first run captures PostgreSQL and TDLib session state. State clearly that STL binaries stay in Telegram, and that restored PostgreSQL metadata and mappings are what allow normal lookup and delivery after restore.
|
||||
|
||||
- [ ] **Step 4: Document monitoring, retention, restore, and the monthly recovery check**
|
||||
|
||||
Document how to inspect timer status, service failures, Restic snapshots, repository checks, and the four restore modes. Explicitly state that `restore-live` is destructive and requires the confirmation flag. Assign the deployment operator a recurring monthly runbook task: run `docker compose --profile backup run --rm backup check --read-data`, then perform the documented disposable restore rehearsal using a selected snapshot. Record the date, snapshot ID, full-check result, restore/health result, and cleanup result. This is an operator-owned manual procedure, not a second production timer or a change to the nightly backup service. Limit the rehearsal to the PostgreSQL logical dump, `tdlib_state`, and `tdlib_bot_state`; do not add `manual_uploads`, STL-binary, archive-content, or channel-forwarding checks. Explain that channel-forwarding behavior and archive/STL-content integrity validation are future work, not restore checks.
|
||||
|
||||
- [ ] **Step 5: Add a concise production-backup section to the root README**
|
||||
|
||||
Add a link from the deployment/operations section to `scripts/backup/README.md`, state that Docker volumes are not backups, and identify the PostgreSQL logical dump and Telegram session volumes as the protected data set. State that manual uploads and temporary ZIPs are excluded and STL binaries remain in Telegram.
|
||||
|
||||
- [ ] **Step 6: Commit documentation**
|
||||
|
||||
```bash
|
||||
git add scripts/backup/README.md README.md
|
||||
git commit -m "docs: document Synology backup and recovery"
|
||||
```
|
||||
|
||||
### Task 6: Verify backup, failure recovery, retention, and restore
|
||||
|
||||
**Files:**
|
||||
- Modify: `scripts/backup/README.md` only if verification commands need correction.
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: the complete backup stack from Tasks 1-5.
|
||||
- Produces: evidence that the acceptance criteria are met, including a full `restic check --read-data`, a disposable restore rehearsal, and a failure-path result. After deployment, the same full-check and rehearsal are an operator-owned monthly runbook task documented in Task 5.
|
||||
|
||||
- [ ] **Step 1: Validate configuration and scripts**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
docker compose --profile backup config --quiet
|
||||
bash -n scripts/backup/container-entrypoint.sh scripts/backup/run-backup.sh scripts/backup/restore.sh
|
||||
systemd-analyze verify deploy/systemd/dragons-stash-backup.service deploy/systemd/dragons-stash-backup.timer
|
||||
```
|
||||
|
||||
Expected: all commands exit `0`.
|
||||
|
||||
- [ ] **Step 2: Seed recognizable database metadata**
|
||||
|
||||
Using the existing app/database workflow, identify a record whose Telegram archive, message, package, and file metadata can be recognized after restore. Record the expected database identifiers before backup. Do not create or retain a local STL binary for this verification.
|
||||
|
||||
- [ ] **Step 3: Run a real backup and inspect the snapshot**
|
||||
|
||||
Run the systemd service manually, then inspect:
|
||||
|
||||
```bash
|
||||
sudo systemctl start dragons-stash-backup.service
|
||||
sudo journalctl -u dragons-stash-backup.service --since "10 minutes ago" --no-pager
|
||||
docker compose --profile backup run --rm backup snapshots
|
||||
docker compose --profile backup run --rm backup check
|
||||
```
|
||||
|
||||
Expected: the service succeeds, the snapshot exists, the repository check succeeds, and all services are running again.
|
||||
|
||||
- [ ] **Step 4: Test the failure path with the NAS unavailable**
|
||||
|
||||
Temporarily unmount the Synology share in a controlled maintenance session, run the systemd service, and confirm it fails before creating a new snapshot. Remount the share and confirm the previously successful snapshot remains listed. Verify that services are running after the failed attempt.
|
||||
|
||||
- [ ] **Step 5: Run the full-read integrity check and rehearse a disposable restore**
|
||||
|
||||
Run `docker compose --profile backup run --rm backup check --read-data` against the selected repository, then restore the selected snapshot to a disposable Compose project or isolated Docker volumes. Import the database dump, restore the two TDLib session trees, start the disposable app/worker/bot services, and call `/api/health`. Confirm the recognizable database metadata and Telegram mappings match the pre-backup record. Record the check and rehearsal evidence as the initial monthly-runbook baseline. Do not assert the presence, checksum, content, or forwarding behavior of STL binaries.
|
||||
|
||||
- [ ] **Step 6: Verify retention behavior**
|
||||
|
||||
Use a disposable repository or controlled test timestamps to create more than 30 daily snapshots, run the retention command after a successful backup, and confirm that the latest 30 daily snapshots remain. Confirm a failed backup does not invoke pruning.
|
||||
|
||||
- [ ] **Step 7: Record verification evidence**
|
||||
|
||||
Add the actual commands, dates, snapshot ID, restore result, and any environment-specific caveats to the operational notes. Do not commit passwords, session contents, database dumps, or NAS addresses that are intended to remain private.
|
||||
|
||||
- [ ] **Step 8: Commit any documentation corrections**
|
||||
|
||||
```bash
|
||||
git add scripts/backup/README.md
|
||||
git commit -m "test: document verified backup and restore procedure"
|
||||
```
|
||||
|
||||
## Plan Self-Review
|
||||
|
||||
- **Spec coverage:** PostgreSQL logical dump, both Telegram session volumes, Synology NFS, Restic encryption, 30-day retention, maintenance window, service restart on failure, guarded restore, and an explicitly deployment-operator-owned monthly `restic check --read-data` plus disposable restore rehearsal are covered by Tasks 1-6. The initial run is verified in Task 6 and the recurring runbook is documented in Task 5; neither adds a second production timer.
|
||||
- **Exclusions:** `manual_uploads` and `tmp_zips` are excluded; local STL retention, restored STL binaries, file-path/checksum validation, channel forwarding, and archive/STL-content integrity checks are not implementation requirements.
|
||||
- **Placeholder scan:** No `TBD`, `TODO`, or unspecified implementation task remains. Environment-dependent values are explicit configuration variables or operator-supplied paths.
|
||||
- **Type/interface consistency:** The Compose service name is consistently `backup`; the container command modes are `backup` and `restore`; the host wrapper owns service lifecycle; the restore script owns destructive confirmation; Restic owns snapshots and pruning.
|
||||
- **Scope check:** The plan contains one operational subsystem with separate backup, restore, scheduling, and documentation units that can each be reviewed and tested independently.
|
||||
@@ -0,0 +1,555 @@
|
||||
# NAS Backup for Postgres + TDLib State Implementation Plan
|
||||
|
||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
||||
|
||||
**Goal:** Add a `backup` container to the DragonsStash stack that takes daily, encrypted, deduplicated backups of the Postgres database and both TDLib state volumes, and ships them to a Synology NAS over NFS.
|
||||
|
||||
**Architecture:** A small Alpine-based image (restic + postgresql16-client + curl + dcron) runs as its own compose service. A crontab fires `backup.sh` daily at 03:00, which dumps Postgres, tars the TDLib volumes, hands both to `restic backup` against an NFS-backed Docker volume, prunes with `restic forget --keep-daily 14`, and reports success/failure to an Uptime Kuma push monitor. Matches the existing `worker`/`bot` pattern: build context in the repo's `docker-compose.yml`, prebuilt image in `/opt/stacks/DragonsStash/docker-compose.yml`, built and pushed by `.drone.yml`.
|
||||
|
||||
**Tech Stack:** Alpine 3.20, restic 0.16, postgresql16-client, dcron, bash, Docker Compose NFS volume driver.
|
||||
|
||||
## Global Constraints
|
||||
|
||||
- Retention: `restic forget --keep-daily 14 --prune` (14-day window, per approved spec).
|
||||
- Schedule: daily backup at 03:00, weekly `restic check` at 04:00 Sunday.
|
||||
- No Docker socket mount, no `privileged: true` — the backup container must not be able to control sibling containers.
|
||||
- No host-level mount — NFS access only via Docker's native `driver_opts: type: nfs` volume, never `/etc/fstab`.
|
||||
- Encryption and retention are restic's job — no hand-rolled `age`/`gpg`/`find -mtime` logic.
|
||||
- TDLib volumes are tarred live (best-effort) — never pause `worker`/`bot` for the backup.
|
||||
- Restore is a manual, documented procedure only — never scripted/automated.
|
||||
|
||||
**Required user input before Task 5 can run:** `NAS_HOST` and `NAS_EXPORT_PATH` (the Synology NFS share details) and a Kuma Push-monitor URL (`KUMA_PUSH_URL`, created manually in the existing Uptime Kuma instance, ~26h expected heartbeat interval). Tasks 1–4 need none of these and can proceed immediately; do not substitute placeholder values for them in Task 5 — stop and ask the user instead.
|
||||
|
||||
---
|
||||
|
||||
### Task 1: Backup image (Dockerfile + entrypoint)
|
||||
|
||||
**Files:**
|
||||
- Create: `backup/Dockerfile`
|
||||
- Create: `backup/entrypoint.sh`
|
||||
- Create: `backup/crontab`
|
||||
|
||||
**Interfaces:**
|
||||
- Produces: a buildable image tagged `dragonsstash-backup:test` locally, with `/entrypoint.sh` as `ENTRYPOINT`, `/backup.sh` present at the image root (written in Task 2 — this task only needs the `COPY` line and a placeholder-free stub isn't acceptable, so create an empty‑body-but-real `backup/backup.sh` here containing just `#!/bin/bash` + `exit 0`, and Task 2 replaces its contents), `restic`, `pg_dump`/`pg_restore`, `curl`, `tar`, `bash`, `dcron` all on `PATH`.
|
||||
- Consumes: nothing from earlier tasks.
|
||||
|
||||
- [ ] **Step 1: Write `backup/backup.sh` stub**
|
||||
|
||||
```bash
|
||||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
exit 0
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Write `backup/entrypoint.sh`**
|
||||
|
||||
```bash
|
||||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
|
||||
if ! restic snapshots >/dev/null 2>&1; then
|
||||
restic init
|
||||
fi
|
||||
|
||||
exec crond -f -l 2
|
||||
```
|
||||
|
||||
- [ ] **Step 3: Write `backup/crontab`**
|
||||
|
||||
```
|
||||
0 3 * * * /backup.sh >> /proc/1/fd/1 2>&1
|
||||
0 4 * * 0 restic check >> /proc/1/fd/1 2>&1
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Write `backup/Dockerfile`**
|
||||
|
||||
```dockerfile
|
||||
FROM alpine:3.20
|
||||
|
||||
RUN apk add --no-cache restic postgresql16-client curl tzdata dcron tar bash
|
||||
|
||||
COPY backup/backup.sh /backup.sh
|
||||
COPY backup/entrypoint.sh /entrypoint.sh
|
||||
COPY backup/crontab /etc/crontabs/root
|
||||
|
||||
RUN chmod +x /backup.sh /entrypoint.sh
|
||||
|
||||
ENTRYPOINT ["/entrypoint.sh"]
|
||||
```
|
||||
|
||||
- [ ] **Step 5: Build the image**
|
||||
|
||||
Run: `cd /home/sam/Documents/DragonsStash && docker build -t dragonsstash-backup:test -f backup/Dockerfile .`
|
||||
Expected: build completes with `Successfully tagged dragonsstash-backup:test` (or Buildkit's equivalent final `naming to docker.io/library/dragonsstash-backup:test done`), no errors.
|
||||
|
||||
- [ ] **Step 6: Verify the tools are present**
|
||||
|
||||
Run: `docker run --rm dragonsstash-backup:test restic version && docker run --rm dragonsstash-backup:test pg_dump --version`
|
||||
Expected: `restic 0.16.x ...` and `pg_dump (PostgreSQL) 16.x` printed, both commands exit 0.
|
||||
|
||||
- [ ] **Step 7: Verify the entrypoint initializes an empty repo and starts cron**
|
||||
|
||||
```bash
|
||||
mkdir -p /tmp/backup-repo-smoke
|
||||
docker run -d --name backup-smoke \
|
||||
-e RESTIC_REPOSITORY=/backups/restic-repo -e RESTIC_PASSWORD=smoketest \
|
||||
-v /tmp/backup-repo-smoke:/backups \
|
||||
dragonsstash-backup:test
|
||||
sleep 2
|
||||
docker logs backup-smoke
|
||||
docker exec backup-smoke restic snapshots
|
||||
docker rm -f backup-smoke
|
||||
rm -rf /tmp/backup-repo-smoke
|
||||
```
|
||||
|
||||
Expected: `docker logs` shows no errors (restic init ran silently); `restic snapshots` prints an empty snapshot list (repo exists, header row only, no error).
|
||||
|
||||
- [ ] **Step 8: Commit**
|
||||
|
||||
```bash
|
||||
git add backup/Dockerfile backup/entrypoint.sh backup/backup.sh backup/crontab
|
||||
git commit -m "Add backup service image (Dockerfile, entrypoint, crontab)"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 2: `backup.sh` script
|
||||
|
||||
**Files:**
|
||||
- Modify: `backup/backup.sh` (replace Task 1's stub with the real script)
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: the image built in Task 1 (`dragonsstash-backup:test`), rebuilt after this change.
|
||||
- Produces: `/backup.sh`, invoked by cron in Task 1's `crontab` and manually in Task 6's verification. Reads env vars `POSTGRES_USER`, `PGPASSWORD`, `POSTGRES_DB`, `RESTIC_REPOSITORY`, `RESTIC_PASSWORD`, `KUMA_PUSH_URL`. Assumes network hostname `dragonsstash-db:5432` for Postgres and mounts `/data/tdlib-worker`, `/data/tdlib-bot` (read-only) for TDLib state.
|
||||
|
||||
- [ ] **Step 1: Replace `backup/backup.sh` with the real script**
|
||||
|
||||
```bash
|
||||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
|
||||
report_failure() {
|
||||
curl -fsS "$KUMA_PUSH_URL" --get \
|
||||
--data-urlencode "status=down" \
|
||||
--data-urlencode "msg=$BASH_COMMAND failed" || true
|
||||
}
|
||||
trap report_failure ERR
|
||||
|
||||
DUMP_FILE=/tmp/dragonsstash.dump
|
||||
TAR_FILE=/tmp/tdlib.tar.gz
|
||||
|
||||
pg_dump -h dragonsstash-db -U "$POSTGRES_USER" -d "$POSTGRES_DB" -Fc -f "$DUMP_FILE"
|
||||
|
||||
tar czf "$TAR_FILE" -C /data tdlib-worker tdlib-bot
|
||||
|
||||
restic backup "$DUMP_FILE" "$TAR_FILE"
|
||||
restic forget --keep-daily 14 --prune
|
||||
|
||||
rm -f "$DUMP_FILE" "$TAR_FILE"
|
||||
|
||||
curl -fsS "$KUMA_PUSH_URL" --get \
|
||||
--data-urlencode "status=up" \
|
||||
--data-urlencode "msg=OK"
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Rebuild the image**
|
||||
|
||||
Run: `docker build -t dragonsstash-backup:test -f backup/Dockerfile .`
|
||||
Expected: build succeeds.
|
||||
|
||||
- [ ] **Step 3: Stand up a scratch Postgres aliased as `dragonsstash-db`**
|
||||
|
||||
```bash
|
||||
docker network create backup-test-net 2>/dev/null || true
|
||||
docker run -d --name test-pg --network backup-test-net --network-alias dragonsstash-db \
|
||||
-e POSTGRES_USER=dragons -e POSTGRES_PASSWORD=stash -e POSTGRES_DB=dragonsstash \
|
||||
postgres:16-alpine
|
||||
sleep 5
|
||||
docker exec test-pg pg_isready -U dragons -d dragonsstash
|
||||
```
|
||||
|
||||
Expected: `accepting connections`.
|
||||
|
||||
- [ ] **Step 4: Stand up a mock Kuma push endpoint**
|
||||
|
||||
```bash
|
||||
mkdir -p /tmp/mock-kuma-root && touch /tmp/mock-kuma-root/push
|
||||
docker run -d --name mock-kuma --network backup-test-net \
|
||||
-v /tmp/mock-kuma-root:/srv -w /srv python:3-alpine \
|
||||
python3 -m http.server 8000
|
||||
sleep 1
|
||||
```
|
||||
|
||||
- [ ] **Step 5: Prepare fake TDLib state and a local restic repo dir**
|
||||
|
||||
```bash
|
||||
mkdir -p /tmp/backup-test/tdlib-worker /tmp/backup-test/tdlib-bot /tmp/backup-test/repo
|
||||
echo "fake-session" > /tmp/backup-test/tdlib-worker/state.bin
|
||||
echo "fake-session" > /tmp/backup-test/tdlib-bot/state.bin
|
||||
```
|
||||
|
||||
- [ ] **Step 6: Initialize the test restic repo and run `backup.sh` (success path)**
|
||||
|
||||
```bash
|
||||
docker run --rm --network backup-test-net \
|
||||
-e RESTIC_REPOSITORY=/backups/restic-repo -e RESTIC_PASSWORD=testpassword \
|
||||
-v /tmp/backup-test/repo:/backups \
|
||||
--entrypoint restic dragonsstash-backup:test init
|
||||
|
||||
docker run --rm --network backup-test-net \
|
||||
-e POSTGRES_USER=dragons -e POSTGRES_PASSWORD=stash -e PGPASSWORD=stash -e POSTGRES_DB=dragonsstash \
|
||||
-e RESTIC_REPOSITORY=/backups/restic-repo -e RESTIC_PASSWORD=testpassword \
|
||||
-e KUMA_PUSH_URL=http://mock-kuma:8000/push \
|
||||
-v /tmp/backup-test/tdlib-worker:/data/tdlib-worker:ro \
|
||||
-v /tmp/backup-test/tdlib-bot:/data/tdlib-bot:ro \
|
||||
-v /tmp/backup-test/repo:/backups \
|
||||
--entrypoint /backup.sh dragonsstash-backup:test
|
||||
echo "exit code: $?"
|
||||
```
|
||||
|
||||
Expected: exit code `0`, restic prints a line like `snapshot xxxxxxxx saved`, no error output.
|
||||
|
||||
- [ ] **Step 7: Verify the snapshot landed and contains both files**
|
||||
|
||||
```bash
|
||||
docker run --rm -v /tmp/backup-test/repo:/backups \
|
||||
-e RESTIC_REPOSITORY=/backups/restic-repo -e RESTIC_PASSWORD=testpassword \
|
||||
--entrypoint restic dragonsstash-backup:test snapshots
|
||||
|
||||
docker run --rm -v /tmp/backup-test/repo:/backups \
|
||||
-e RESTIC_REPOSITORY=/backups/restic-repo -e RESTIC_PASSWORD=testpassword \
|
||||
--entrypoint restic dragonsstash-backup:test ls latest
|
||||
```
|
||||
|
||||
Expected: `snapshots` shows exactly one entry; `ls latest` lists `/tmp/dragonsstash.dump` and `/tmp/tdlib.tar.gz`.
|
||||
|
||||
- [ ] **Step 8: Verify the failure path reports to Kuma**
|
||||
|
||||
```bash
|
||||
docker run --rm --network backup-test-net \
|
||||
-e POSTGRES_USER=dragons -e POSTGRES_PASSWORD=wrongpass -e PGPASSWORD=wrongpass -e POSTGRES_DB=dragonsstash \
|
||||
-e RESTIC_REPOSITORY=/backups/restic-repo -e RESTIC_PASSWORD=testpassword \
|
||||
-e KUMA_PUSH_URL=http://mock-kuma:8000/push \
|
||||
-v /tmp/backup-test/tdlib-worker:/data/tdlib-worker:ro \
|
||||
-v /tmp/backup-test/tdlib-bot:/data/tdlib-bot:ro \
|
||||
-v /tmp/backup-test/repo:/backups \
|
||||
--entrypoint /backup.sh dragonsstash-backup:test
|
||||
echo "exit code: $?"
|
||||
docker logs mock-kuma | tail -5
|
||||
```
|
||||
|
||||
Expected: exit code nonzero (pg_dump auth failure trips `set -e`); `docker logs mock-kuma` shows a GET request line containing `status=down`.
|
||||
|
||||
- [ ] **Step 9: Clean up test resources**
|
||||
|
||||
```bash
|
||||
docker rm -f test-pg mock-kuma
|
||||
docker network rm backup-test-net
|
||||
rm -rf /tmp/backup-test /tmp/mock-kuma-root
|
||||
```
|
||||
|
||||
- [ ] **Step 10: Commit**
|
||||
|
||||
```bash
|
||||
git add backup/backup.sh
|
||||
git commit -m "Implement backup.sh: pg_dump + tdlib tar + restic backup/forget + Kuma reporting"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 3: Wire the `backup` service into the repo's `docker-compose.yml`
|
||||
|
||||
**Files:**
|
||||
- Modify: `docker-compose.yml`
|
||||
- Modify: `.env.example`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `backup/Dockerfile` (Task 1), `backup/backup.sh` (Task 2).
|
||||
- Produces: a `backup` compose service buildable via `docker compose build backup`, and a `nas_backups` named volume other tasks (5) will mirror into the production compose file.
|
||||
|
||||
- [ ] **Step 1: Add the `backup` service to `docker-compose.yml`**
|
||||
|
||||
Insert after the existing `bot` service (before `db`):
|
||||
|
||||
```yaml
|
||||
backup:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: backup/Dockerfile
|
||||
pull_policy: never
|
||||
environment:
|
||||
- POSTGRES_USER=${POSTGRES_USER:-dragons}
|
||||
- POSTGRES_PASSWORD=${POSTGRES_PASSWORD:-stash}
|
||||
- PGPASSWORD=${POSTGRES_PASSWORD:-stash}
|
||||
- POSTGRES_DB=${POSTGRES_DB:-dragonsstash}
|
||||
- RESTIC_REPOSITORY=/backups/restic-repo
|
||||
- RESTIC_PASSWORD=${RESTIC_PASSWORD:?Set RESTIC_PASSWORD in .env}
|
||||
- KUMA_PUSH_URL=${KUMA_PUSH_URL:?Set KUMA_PUSH_URL in .env}
|
||||
- TZ=${TZ:-Etc/UTC}
|
||||
volumes:
|
||||
- tdlib_state:/data/tdlib-worker:ro
|
||||
- tdlib_bot_state:/data/tdlib-bot:ro
|
||||
- nas_backups:/backups
|
||||
depends_on:
|
||||
db:
|
||||
condition: service_healthy
|
||||
restart: unless-stopped
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 256M
|
||||
networks:
|
||||
- backend
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Add the `nas_backups` volume to the `volumes:` block**
|
||||
|
||||
```yaml
|
||||
nas_backups:
|
||||
driver_opts:
|
||||
type: nfs
|
||||
o: "addr=${NAS_HOST},rw,nfsvers=4,soft,timeo=100"
|
||||
device: ":${NAS_EXPORT_PATH}"
|
||||
```
|
||||
|
||||
- [ ] **Step 3: Document the new env vars in `.env.example`**
|
||||
|
||||
Append:
|
||||
|
||||
```
|
||||
# Backup (NAS via NFS + restic)
|
||||
NAS_HOST="" # Synology NAS IP or hostname reachable from this host
|
||||
NAS_EXPORT_PATH="" # NFS export path, e.g. /volume1/dragonsstash-backups
|
||||
RESTIC_PASSWORD="" # generate with: openssl rand -base64 32
|
||||
KUMA_PUSH_URL="" # Uptime Kuma Push monitor URL (create the monitor first)
|
||||
TZ="Etc/UTC"
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Validate the compose file parses**
|
||||
|
||||
Run (with dummy values so the `:?` guards don't fail parsing — `AUTH_SECRET` is required by the existing `app` service, not by this change, but `config` validates the whole file):
|
||||
```bash
|
||||
RESTIC_PASSWORD=dummy KUMA_PUSH_URL=http://dummy NAS_HOST=dummy NAS_EXPORT_PATH=/dummy AUTH_SECRET=dummy \
|
||||
docker compose -f docker-compose.yml config --quiet
|
||||
```
|
||||
Expected: no output, exit code 0 (a syntax/interpolation error would print to stderr and exit nonzero).
|
||||
|
||||
- [ ] **Step 5: Validate the service actually builds through Compose**
|
||||
|
||||
Run:
|
||||
```bash
|
||||
RESTIC_PASSWORD=dummy KUMA_PUSH_URL=http://dummy NAS_HOST=dummy NAS_EXPORT_PATH=/dummy AUTH_SECRET=dummy \
|
||||
docker compose -f docker-compose.yml build backup
|
||||
```
|
||||
Expected: build succeeds (reuses Task 1's image layers).
|
||||
|
||||
- [ ] **Step 6: Commit**
|
||||
|
||||
```bash
|
||||
git add docker-compose.yml .env.example
|
||||
git commit -m "Add backup service and nas_backups volume to docker-compose.yml"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 4: CI — build and push the backup image
|
||||
|
||||
**Files:**
|
||||
- Modify: `.drone.yml`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `backup/Dockerfile` (Task 1).
|
||||
- Produces: `git.samagsteribbe.nl/admin/dragonsstash-backup:latest` (and `:<short-sha>`), pushed on every push to `main`. Task 5's production compose file references this image tag.
|
||||
|
||||
- [ ] **Step 1: Add a `build-backup` step, mirroring `build-worker`/`build-bot`**
|
||||
|
||||
Insert after the existing `build-bot` step in `.drone.yml`:
|
||||
|
||||
```yaml
|
||||
- name: build-backup
|
||||
image: plugins/docker
|
||||
depends_on: [clone]
|
||||
settings:
|
||||
repo: git.samagsteribbe.nl/admin/dragonsstash-backup
|
||||
registry: git.samagsteribbe.nl
|
||||
dockerfile: backup/Dockerfile
|
||||
tags:
|
||||
- latest
|
||||
- "${DRONE_COMMIT_SHA:0:8}"
|
||||
username:
|
||||
from_secret: gitea_username
|
||||
password:
|
||||
from_secret: gitea_password
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Add `build-backup` to the `deploy` step's `depends_on`**
|
||||
|
||||
Change:
|
||||
```yaml
|
||||
- name: deploy
|
||||
image: alpine
|
||||
depends_on: [build-app, build-worker, build-bot]
|
||||
```
|
||||
to:
|
||||
```yaml
|
||||
- name: deploy
|
||||
image: alpine
|
||||
depends_on: [build-app, build-worker, build-bot, build-backup]
|
||||
```
|
||||
|
||||
- [ ] **Step 3: Validate YAML syntax**
|
||||
|
||||
Run: `python3 -c "import yaml; yaml.safe_load(open('.drone.yml')); print('OK')"`
|
||||
Expected: `OK` printed, no exception.
|
||||
|
||||
- [ ] **Step 4: Commit**
|
||||
|
||||
```bash
|
||||
git add .drone.yml
|
||||
git commit -m "Add build-backup CI step, include it in deploy dependencies"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 5: Wire production (`/opt/stacks/DragonsStash`) — requires NAS + Kuma details from the user
|
||||
|
||||
**Do not start this task until the user has supplied `NAS_HOST` and `NAS_EXPORT_PATH` for the Synology NFS share, and has created an Uptime Kuma Push monitor (name it `dragonsstash-backup`, ~26h expected heartbeat interval) and shared its push URL. If any of these are missing, stop and ask — do not substitute placeholder values here, since this file drives the real deployment.**
|
||||
|
||||
**Files:**
|
||||
- Modify: `/opt/stacks/DragonsStash/docker-compose.yml`
|
||||
- Modify: `/opt/stacks/DragonsStash/.env`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `git.samagsteribbe.nl/admin/dragonsstash-backup:latest` (published by Task 4's CI step once merged/pushed), the real `NAS_HOST`/`NAS_EXPORT_PATH`/`KUMA_PUSH_URL` values gathered above.
|
||||
- Produces: a running `dragonsstash-backup` container on the production host, verified in Task 6.
|
||||
|
||||
- [ ] **Step 1: Generate `RESTIC_PASSWORD` and add all four new vars to `/opt/stacks/DragonsStash/.env`**
|
||||
|
||||
```bash
|
||||
cd /opt/stacks/DragonsStash
|
||||
printf '\n# Backup (NAS via NFS + restic)\nNAS_HOST="<value from user>"\nNAS_EXPORT_PATH="<value from user>"\nRESTIC_PASSWORD="%s"\nKUMA_PUSH_URL="<value from user>"\nTZ="Etc/UTC"\n' "$(openssl rand -base64 32)" >> .env
|
||||
```
|
||||
|
||||
Replace the two `<value from user>` placeholders with the real NAS details and Kuma push URL before saving — this step cannot be completed with the literal placeholder text left in place.
|
||||
|
||||
- [ ] **Step 2: Add the `backup` service to `/opt/stacks/DragonsStash/docker-compose.yml`**
|
||||
|
||||
Insert after the existing `bot` service (before `db`):
|
||||
|
||||
```yaml
|
||||
backup:
|
||||
image: git.samagsteribbe.nl/admin/dragonsstash-backup:latest
|
||||
container_name: dragonsstash-backup
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
- POSTGRES_USER=${POSTGRES_USER:-dragons}
|
||||
- POSTGRES_PASSWORD=${POSTGRES_PASSWORD:-stash}
|
||||
- PGPASSWORD=${POSTGRES_PASSWORD:-stash}
|
||||
- POSTGRES_DB=${POSTGRES_DB:-dragonsstash}
|
||||
- RESTIC_REPOSITORY=/backups/restic-repo
|
||||
- RESTIC_PASSWORD=${RESTIC_PASSWORD:?Set RESTIC_PASSWORD in .env}
|
||||
- KUMA_PUSH_URL=${KUMA_PUSH_URL:?Set KUMA_PUSH_URL in .env}
|
||||
- TZ=${TZ:-Etc/UTC}
|
||||
volumes:
|
||||
- tdlib_state:/data/tdlib-worker:ro
|
||||
- tdlib_bot_state:/data/tdlib-bot:ro
|
||||
- nas_backups:/backups
|
||||
depends_on:
|
||||
db:
|
||||
condition: service_healthy
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 256M
|
||||
networks:
|
||||
- internal
|
||||
```
|
||||
|
||||
- [ ] **Step 3: Add the `nas_backups` volume**
|
||||
|
||||
```yaml
|
||||
nas_backups:
|
||||
driver_opts:
|
||||
type: nfs
|
||||
o: "addr=${NAS_HOST},rw,nfsvers=4,soft,timeo=100"
|
||||
device: ":${NAS_EXPORT_PATH}"
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Validate the production compose file parses with the real `.env`**
|
||||
|
||||
Run: `cd /opt/stacks/DragonsStash && docker compose config --quiet`
|
||||
Expected: no output, exit code 0.
|
||||
|
||||
- [ ] **Step 5: Commit is not applicable here** — `/opt/stacks/DragonsStash` is a deployed copy, not the git repo (confirm with `git -C /opt/stacks/DragonsStash status` — expect "not a git repository"). Skip committing; Task 6 deploys these file changes directly.
|
||||
|
||||
---
|
||||
|
||||
### Task 6: Deploy and verify
|
||||
|
||||
**Files:** none (operational task)
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: everything from Tasks 1–5.
|
||||
- Produces: a running, verified backup on the real NAS, and one completed restore drill.
|
||||
|
||||
- [ ] **Step 1: Confirm with the user before pushing/deploying**
|
||||
|
||||
Pushing to `main` triggers Drone CI to build all four images and deploy to the production host via SSH (`docker compose pull && docker compose up -d`). Confirm the user wants this to happen now before proceeding — this affects the live stack.
|
||||
|
||||
- [ ] **Step 2: Push to `main`**
|
||||
|
||||
```bash
|
||||
cd /home/sam/Documents/DragonsStash
|
||||
git push origin main
|
||||
```
|
||||
|
||||
Expected: Drone pipeline runs `build-app`, `build-worker`, `build-bot`, `build-backup`, then `deploy`, all green. Check via the Drone UI or `drone build info admin/DragonsStash <build-number>` if the `drone` CLI is configured.
|
||||
|
||||
- [ ] **Step 3: Confirm the container is up on the production host**
|
||||
|
||||
```bash
|
||||
ssh sam@192.168.68.68 "docker ps --filter name=dragonsstash-backup --format '{{.Names}}\t{{.Status}}'"
|
||||
```
|
||||
|
||||
Expected: `dragonsstash-backup Up ...`.
|
||||
|
||||
- [ ] **Step 4: Trigger one manual backup run and confirm a snapshot lands on the NAS**
|
||||
|
||||
```bash
|
||||
ssh sam@192.168.68.68 "docker exec dragonsstash-backup /backup.sh"
|
||||
ssh sam@192.168.68.68 "docker exec dragonsstash-backup restic snapshots"
|
||||
```
|
||||
|
||||
Expected: `backup.sh` exits 0; `restic snapshots` lists exactly one entry.
|
||||
|
||||
- [ ] **Step 5: Confirm the Uptime Kuma push monitor shows green**
|
||||
|
||||
Open the Uptime Kuma dashboard and check the `dragonsstash-backup` monitor's status is up with a recent heartbeat.
|
||||
|
||||
- [ ] **Step 6: Restore drill — prove the backup is actually restorable**
|
||||
|
||||
```bash
|
||||
ssh sam@192.168.68.68 "docker exec dragonsstash-backup restic restore latest --target /tmp/restore-drill"
|
||||
ssh sam@192.168.68.68 "docker exec dragonsstash-backup ls -la /tmp/restore-drill/tmp"
|
||||
```
|
||||
|
||||
Expected: `/tmp/restore-drill/tmp/dragonsstash.dump` and `/tmp/restore-drill/tmp/tdlib.tar.gz` both present with nonzero size. Then, on a scratch Postgres (not the live `dragonsstash-db`), confirm the dump restores cleanly:
|
||||
|
||||
```bash
|
||||
ssh sam@192.168.68.68 "docker run -d --rm --name restore-drill-pg --network dragonsstash_internal \
|
||||
-e POSTGRES_USER=drill -e POSTGRES_PASSWORD=drill -e POSTGRES_DB=drill postgres:16-alpine"
|
||||
ssh sam@192.168.68.68 "docker cp dragonsstash-backup:/tmp/restore-drill/tmp/dragonsstash.dump /tmp/dragonsstash.dump"
|
||||
ssh sam@192.168.68.68 "docker cp /tmp/dragonsstash.dump restore-drill-pg:/tmp/dragonsstash.dump"
|
||||
ssh sam@192.168.68.68 "docker exec -e PGPASSWORD=drill restore-drill-pg pg_restore -U drill -d drill --clean --if-exists /tmp/dragonsstash.dump"
|
||||
ssh sam@192.168.68.68 "docker exec -e PGPASSWORD=drill restore-drill-pg psql -U drill -d drill -c '\\dt' | head -20"
|
||||
ssh sam@192.168.68.68 "docker rm -f restore-drill-pg"
|
||||
```
|
||||
|
||||
Expected: `pg_restore` completes without fatal errors; `\dt` lists the app's tables (e.g. `Package`, `User`, `TelegramLink`).
|
||||
|
||||
- [ ] **Step 7: Clean up drill artifacts**
|
||||
|
||||
```bash
|
||||
ssh sam@192.168.68.68 "docker exec dragonsstash-backup rm -rf /tmp/restore-drill"
|
||||
ssh sam@192.168.68.68 "rm -f /tmp/dragonsstash.dump"
|
||||
```
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,125 @@
|
||||
# Database and File Backup Design
|
||||
|
||||
**Date:** 2026-07-21
|
||||
**Status:** Design approved for written-spec review
|
||||
**Scope:** Disaster recovery for the Docker Compose deployment
|
||||
|
||||
## Problem
|
||||
|
||||
Dragon's Stash currently persists PostgreSQL in a Docker named volume. The Telegram worker and bot also persist authentication and session state in the `tdlib_state` and `tdlib_bot_state` volumes. Docker volumes protect against container recreation, but they are not off-host backups. A host disk failure, accidental deletion, corruption, or ransomware event could destroy the database and require both Telegram clients to authenticate again.
|
||||
|
||||
STL binaries are intentionally not retained as a local recovery set. They remain in Telegram. The PostgreSQL database preserves the archive, message, package, and file metadata needed to locate and send those Telegram-hosted binaries after the application database is restored. Existing worker cleanup behavior is unchanged.
|
||||
|
||||
## Goals
|
||||
|
||||
- Protect PostgreSQL data against loss of the application host with a logical backup.
|
||||
- Preserve Telegram worker and bot session state so a host restore does not normally require re-authentication.
|
||||
- Store backups on a Synology NAS over an authenticated, host-restricted NFS share.
|
||||
- Create one recoverable snapshot containing related database and session state.
|
||||
- Retain 30 daily recovery points.
|
||||
- Provide a documented, repeatable, guarded restore process.
|
||||
- Detect failed or corrupt backups instead of silently pruning the last good copy.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- Building an in-app backup-management UI.
|
||||
- Backing up `manual_uploads`, retaining completed STL binaries locally, or changing worker cleanup behavior.
|
||||
- Backing up temporary ZIP processing data in `tmp_zips`.
|
||||
- Copying the raw `postgres_data` volume as the primary database backup.
|
||||
- Implementing future Telegram channel-forwarding behavior.
|
||||
- Validating the content or binary integrity of Telegram archives or STL files. This is future work and is outside the backup/restore feature.
|
||||
- Providing protection against loss of the NAS itself. A later Synology Hyper Backup task can replicate this repository to another device or cloud destination.
|
||||
|
||||
## Selected approach
|
||||
|
||||
Use a Linux-host backup script, scheduled by a systemd timer, with Restic writing to an encrypted repository on a Synology NFS share.
|
||||
|
||||
This approach keeps backup and restore explicit, avoids tying recovery to PostgreSQL's internal data-directory layout, and captures the sensitive TDLib state alongside the database. Restic repository encryption protects database contents and Telegram session state if the NAS share is accessed directly.
|
||||
|
||||
## Storage layout
|
||||
|
||||
### Synology
|
||||
|
||||
Create a dedicated shared folder, for example `dragonsstash-backups`, with:
|
||||
|
||||
- The dedicated backup shared folder is exported through NFS only to the Docker host's fixed IP address.
|
||||
- No Internet exposure.
|
||||
- Sufficient capacity for the repository plus growth and safety margin.
|
||||
|
||||
The Linux host mounts the share at a stable path such as `/mnt/dragonsstash-backups` using systemd-aware network mount options so a NAS outage does not block normal boot indefinitely.
|
||||
|
||||
### Restic repository
|
||||
|
||||
The repository lives below the mounted share. The Restic password is stored separately from the repository in a root-readable host secret file and must also be recorded in the operator's offline password-management system. Losing both the NAS and the only copy of the Restic password makes encrypted backups unrecoverable.
|
||||
|
||||
Each successful Restic snapshot contains:
|
||||
|
||||
- A PostgreSQL custom-format dump generated for that run.
|
||||
- The contents of the `tdlib_state` Docker volume.
|
||||
- The contents of the `tdlib_bot_state` Docker volume.
|
||||
- A small manifest with the backup timestamp, application image/version, database migration state, and captured volume paths.
|
||||
|
||||
The `manual_uploads` and `tmp_zips` volumes are excluded. STL binaries continue to live in Telegram; the restored database supplies the metadata and mappings required for the worker and bot to locate and send them.
|
||||
|
||||
## Backup flow
|
||||
|
||||
The systemd timer invokes one backup command at the chosen nightly time. The command:
|
||||
|
||||
1. Acquires an exclusive lock and refuses to run if another backup is active.
|
||||
2. Verifies that the NFS mount is present, writable, and points to the expected backup directory.
|
||||
3. Stops the `app`, `worker`, and `bot` services while leaving PostgreSQL running.
|
||||
4. Creates a PostgreSQL custom-format dump from the running database.
|
||||
5. Creates a manifest for the backup.
|
||||
6. Runs one Restic backup over the dump and the read-only mounted TDLib session volumes.
|
||||
7. Verifies that Restic created the snapshot successfully.
|
||||
8. Applies the retention policy: keep the latest 30 daily snapshots, then prune unreferenced data.
|
||||
9. Restarts all stopped services, whether the backup succeeded or failed.
|
||||
|
||||
The service-stop window ensures that application writes and TDLib session updates do not occur while the corresponding data is captured. A failed run must never trigger retention pruning.
|
||||
|
||||
## Restore flow
|
||||
|
||||
The restore tooling and documentation will support this sequence:
|
||||
|
||||
1. Stop `app`, `worker`, and `bot` for a live restore.
|
||||
2. Select and inspect a Restic snapshot.
|
||||
3. Restore the PostgreSQL dump and TDLib session contents to a staging location.
|
||||
4. Preserve the current database and volumes or confirm that the operator intends to replace them.
|
||||
5. Restore `tdlib_state` and `tdlib_bot_state` into their Docker volumes with the expected ownership and paths.
|
||||
6. Restore the database from the custom-format dump into the configured PostgreSQL database.
|
||||
7. Verify the dump, session-volume layout, database connectivity, and worker/bot authentication startup state.
|
||||
8. Start the services and inspect logs for startup, migration, and worker/bot authentication errors.
|
||||
|
||||
After restore, existing database metadata and mappings allow normal Telegram-based STL lookup and delivery. Restoring STL binaries, forwarding Telegram content, and checking archive/STL binary integrity are outside this restore flow.
|
||||
|
||||
The restore process must be safe to rehearse against a disposable Compose project without modifying the live deployment.
|
||||
|
||||
## Failure handling and verification
|
||||
|
||||
- A missing or read-only NAS mount fails the backup before services are stopped where possible.
|
||||
- A lock prevents overlapping backups.
|
||||
- A cleanup trap or equivalent guarantees service restart after errors.
|
||||
- Backup failure produces a non-zero systemd result and a clear log entry.
|
||||
- Retention pruning runs only after a verified successful snapshot.
|
||||
- A snapshot listing and repository metadata check run after each backup.
|
||||
- The deployment operator completes and records a monthly operational check: `restic check --read-data` followed by a disposable restore rehearsal. This verifies only the PostgreSQL logical dump and TDLib session-state recovery set; archive/STL-content integrity and future forwarding checks remain out of scope.
|
||||
- The project documentation describes how to inspect the last successful snapshot and how to recover when the NAS is unavailable.
|
||||
|
||||
## Security considerations
|
||||
|
||||
- The NFS share is limited to the Docker host and is not exposed to the Internet.
|
||||
- Restic encryption protects the repository at rest.
|
||||
- PostgreSQL credentials, NAS credentials/rules, and the Restic password are never committed to the repository.
|
||||
- Restore commands must avoid printing database passwords or the Restic password in logs.
|
||||
- Telegram session volumes are included because they are operationally valuable, but they must be treated as secrets.
|
||||
|
||||
## Acceptance criteria
|
||||
|
||||
- A nightly systemd timer creates a Restic snapshot on the Synology share.
|
||||
- A snapshot includes a PostgreSQL logical dump and both Telegram session volumes, while excluding `manual_uploads` and `tmp_zips`.
|
||||
- At least 30 daily recovery points are retained.
|
||||
- Each month, the deployment operator runs and records a full `restic check --read-data` and a disposable restore rehearsal of the PostgreSQL dump plus both TDLib session volumes.
|
||||
- A simulated host-loss restore reconstructs the database and Telegram session state in a disposable Compose environment.
|
||||
- The restored database retains the Telegram metadata and mappings the worker and bot use to locate and send STL binaries that remain in Telegram.
|
||||
- A failed backup leaves services running and preserves the last known-good snapshot.
|
||||
- The restore procedure is documented well enough for an operator to execute without reading the implementation.
|
||||
@@ -0,0 +1,205 @@
|
||||
# NAS Backup for Postgres + TDLib State — Design
|
||||
|
||||
**Date:** 2026-07-23
|
||||
**Status:** Approved for planning
|
||||
|
||||
## Summary
|
||||
|
||||
Add a dedicated `backup` container to the DragonsStash stack that takes daily,
|
||||
encrypted, deduplicated backups of the two things that can't be regenerated —
|
||||
the Postgres database (inventory/STL metadata, users, everything the app
|
||||
manages) and the two TDLib state volumes (Telegram session/auth state for the
|
||||
worker and bot) — and ships them to a Synology NAS over NFS. STL archive
|
||||
contents themselves are explicitly out of scope: they only live on this host
|
||||
temporarily and are not backed up.
|
||||
|
||||
Backups are stored via [restic](https://restic.net/), which provides
|
||||
encryption-at-rest, block-level dedup, and retention pruning natively, so no
|
||||
custom encryption or pruning scripts need to be written or maintained.
|
||||
|
||||
## Context
|
||||
|
||||
Current state (as of this design):
|
||||
|
||||
- Production stack runs from `/opt/stacks/DragonsStash/docker-compose.yml` on
|
||||
this Dockge-managed host, pulling prebuilt images from
|
||||
`git.samagsteribbe.nl`. The `docker-compose.yml` in this repo is the
|
||||
build/dev reference and should be kept in sync.
|
||||
- Named volumes in use: `postgres_data` (Postgres 16 data directory),
|
||||
`tdlib_state` (worker's TDLib session), `tdlib_bot_state` (bot's TDLib
|
||||
session), `tmp_zips` and `manual_uploads` (both transient, explicitly out of
|
||||
scope here).
|
||||
- No backup mechanism, NFS mount, or host cron currently exists anywhere in
|
||||
this deployment.
|
||||
- The host already runs Uptime Kuma (used here for backup alerting) and Loki
|
||||
(container logs are presumably already collected there).
|
||||
|
||||
## Requirements
|
||||
|
||||
1. Daily backup of the Postgres database and both TDLib state volumes.
|
||||
2. Backups stored on a Synology NAS via NFS, not on local disk.
|
||||
3. 14-day retention, oldest snapshots pruned automatically.
|
||||
4. Backups encrypted at rest (Postgres dumps and TDLib session files both
|
||||
contain sensitive material — password hashes, Telegram API secrets, live
|
||||
session state).
|
||||
5. Postgres backups must be transactionally consistent regardless of live app
|
||||
traffic. TDLib state backups are best-effort (see Decisions below) — this
|
||||
is an accepted trade-off, not a defect.
|
||||
6. Alert (via existing Uptime Kuma) if a backup run fails or doesn't happen.
|
||||
7. No new host-level state (no `/etc/fstab` entries, no host crontab) — the
|
||||
backup mechanism should be a container, consistent with how everything
|
||||
else on this host is deployed and versioned.
|
||||
8. No new privileged access — specifically, the backup container must not
|
||||
have Docker socket access or any ability to control sibling containers.
|
||||
|
||||
## Decisions
|
||||
|
||||
- **NFS mounted via Docker's native NFS volume driver** (`driver_opts: type:
|
||||
nfs`), not a host-level mount. Keeps all backup-related state inside the
|
||||
compose file instead of split across host config.
|
||||
- **Restic, not hand-rolled tar+age+find.** Restic already solves encryption,
|
||||
dedup, and retention correctly; hand-rolled scripts would be reinventing
|
||||
that logic with more room for bugs.
|
||||
- **TDLib state is tarred live (best-effort), not paused.** Pausing the
|
||||
worker/bot for a clean snapshot would require mounting the Docker socket
|
||||
into the backup container so it could stop/start sibling containers — a
|
||||
real privilege escalation (a compromised backup container could then
|
||||
control any container on the host). The downside of a best-effort tar is
|
||||
bounded: worst case, a bad TDLib restore means redoing the Telegram SMS
|
||||
auth flow, which is the same outcome as having no backup at all. That
|
||||
bounded, low-severity downside doesn't justify the privilege escalation.
|
||||
- **Fixed-time cron (`crond`), not a sleep-loop.** A `sleep 86400` loop drifts
|
||||
on every container restart; a real crontab entry fires at a fixed time of
|
||||
day regardless of restarts, for negligible extra complexity.
|
||||
- **Restore is manual, not automated.** A script capable of restoring can
|
||||
overwrite live state; that should always require a human deliberately
|
||||
running it, not run unattended.
|
||||
|
||||
## Design
|
||||
|
||||
### New service: `backup`
|
||||
|
||||
Added to both `/opt/stacks/DragonsStash/docker-compose.yml` (production) and
|
||||
this repo's `docker-compose.yml` (dev/build reference).
|
||||
|
||||
- **Image**: custom, `FROM alpine:3.20`, `apk add --no-cache restic
|
||||
postgresql16-client curl tzdata dcron tar bash`. No dependency on app
|
||||
source — independent Dockerfile, e.g. `backup/Dockerfile`.
|
||||
- **Scheduling**: `crond -f` in the foreground as the container's entrypoint,
|
||||
with a crontab installed at build time:
|
||||
```
|
||||
0 3 * * * /backup.sh >> /proc/1/fd/1 2>&1
|
||||
0 4 * * 0 restic check >> /proc/1/fd/1 2>&1
|
||||
```
|
||||
(daily dump/backup at 03:00, weekly repo integrity check at 04:00 Sunday).
|
||||
`restic init` runs once at container startup (entrypoint, before `crond`
|
||||
starts), swallowing the "already initialized" error on subsequent
|
||||
container (re)starts.
|
||||
- **Network**: `internal` only — reaches `dragonsstash-db:5432` for
|
||||
`pg_dump`. No ports exposed.
|
||||
- **Volumes**:
|
||||
- `tdlib_state:/data/tdlib-worker:ro`
|
||||
- `tdlib_bot_state:/data/tdlib-bot:ro`
|
||||
- `nas_backups:/backups`, a named volume defined with:
|
||||
```yaml
|
||||
nas_backups:
|
||||
driver_opts:
|
||||
type: nfs
|
||||
o: "addr=${NAS_HOST},rw,nfsvers=4,soft,timeo=100"
|
||||
device: ":${NAS_EXPORT_PATH}"
|
||||
```
|
||||
- **New `.env` entries**: `NAS_HOST`, `NAS_EXPORT_PATH` (NFS share details),
|
||||
`RESTIC_PASSWORD` (repo encryption key), `KUMA_PUSH_URL` (Uptime Kuma push
|
||||
monitor URL). All four are inputs to gather during implementation, not
|
||||
hardcoded.
|
||||
- `restart: unless-stopped`, no `privileged`, no Docker socket mount.
|
||||
|
||||
### `backup.sh`
|
||||
|
||||
```
|
||||
set -euo pipefail
|
||||
trap 'curl -fsS "$KUMA_PUSH_URL" --get --data-urlencode "status=down" \
|
||||
--data-urlencode "msg=$BASH_COMMAND failed"' ERR
|
||||
|
||||
pg_dump -h dragonsstash-db -U "$POSTGRES_USER" -d "$POSTGRES_DB" \
|
||||
-Fc -f /tmp/dragonsstash.dump
|
||||
|
||||
tar czf /tmp/tdlib.tar.gz -C /data tdlib-worker tdlib-bot
|
||||
|
||||
restic backup /tmp/dragonsstash.dump /tmp/tdlib.tar.gz
|
||||
restic forget --keep-daily 14 --prune
|
||||
|
||||
rm -f /tmp/dragonsstash.dump /tmp/tdlib.tar.gz
|
||||
|
||||
curl -fsS "$KUMA_PUSH_URL" --get --data-urlencode "status=up" \
|
||||
--data-urlencode "msg=OK"
|
||||
```
|
||||
|
||||
`PGPASSWORD` and `RESTIC_REPOSITORY=/backups/restic-repo` are set as
|
||||
container environment variables (from `.env`), not inline in the script.
|
||||
|
||||
### Data flow
|
||||
|
||||
```
|
||||
crond (daily 03:00)
|
||||
→ pg_dump (consistent snapshot via Postgres MVCC) → /tmp/dragonsstash.dump
|
||||
→ tar tdlib_state + tdlib_bot_state (best-effort, live) → /tmp/tdlib.tar.gz
|
||||
→ restic backup (encrypt + dedup) → NFS-mounted repo on Synology NAS
|
||||
→ restic forget --keep-daily 14 --prune
|
||||
→ curl Uptime Kuma push monitor (up on success, down + reason on any failure)
|
||||
```
|
||||
|
||||
### Restore (manual, documented procedure — not scripted/automated)
|
||||
|
||||
```
|
||||
restic -r /backups/restic-repo restore latest --target /tmp/restore
|
||||
pg_restore -h dragonsstash-db -U "$POSTGRES_USER" -d "$POSTGRES_DB" \
|
||||
--clean --if-exists /tmp/restore/tmp/dragonsstash.dump
|
||||
# untar /tmp/restore/tmp/tdlib.tar.gz back into the tdlib_state /
|
||||
# tdlib_bot_state volumes (via a throwaway container mounting both)
|
||||
```
|
||||
|
||||
## Alerting
|
||||
|
||||
- One Uptime Kuma **Push** monitor, created manually in the existing Kuma
|
||||
instance, with an expected heartbeat interval of ~26 hours (slack past the
|
||||
24h schedule so one slow run doesn't false-positive). Whatever notification
|
||||
channels are already configured on that monitor fire automatically — no new
|
||||
alerting integration.
|
||||
- `backup.sh` pushes `status=up` on success and `status=down` (with the
|
||||
failing command in `msg`) on any failure, via the `ERR` trap.
|
||||
- Container logs go to stdout, collected the same way every other container's
|
||||
logs already are on this host.
|
||||
|
||||
## Testing
|
||||
|
||||
This repo has no automated test framework (documented convention: manual
|
||||
testing). For this infra change:
|
||||
|
||||
- After deploy: manually run `docker exec dragonsstash-backup /backup.sh`
|
||||
once, confirm a snapshot appears (`restic snapshots`), confirm the Kuma
|
||||
monitor goes green.
|
||||
- **Restore drill** (once, during setup): actually restore the dump into a
|
||||
scratch Postgres and untar the TDLib archive into scratch volumes, to prove
|
||||
the backup is really restorable. Not automated or recurring for now.
|
||||
|
||||
## Out of scope / non-goals
|
||||
|
||||
- Backing up `tmp_zips` or `manual_uploads` — both transient by design.
|
||||
- Automated/scheduled restore testing.
|
||||
- Backing up any other stack on this host (this design is DragonsStash-only,
|
||||
though the pattern — Docker-native NFS volume + restic — could be reused
|
||||
for other stacks later).
|
||||
- Pausing worker/bot for a guaranteed-consistent TDLib snapshot (see
|
||||
Decisions).
|
||||
|
||||
## Files touched
|
||||
|
||||
- `docker-compose.yml` (this repo) and
|
||||
`/opt/stacks/DragonsStash/docker-compose.yml` (production) — add `backup`
|
||||
service, `nas_backups` volume.
|
||||
- `backup/Dockerfile` — new.
|
||||
- `backup/backup.sh` — new.
|
||||
- `backup/crontab` — new.
|
||||
- `.env.example` / `.env` — add `NAS_HOST`, `NAS_EXPORT_PATH`,
|
||||
`RESTIC_PASSWORD`, `KUMA_PUSH_URL`.
|
||||
@@ -0,0 +1,271 @@
|
||||
# Provenance Backfill on Re-Index — Design
|
||||
|
||||
**Date:** 2026-07-23
|
||||
**Status:** Approved for planning
|
||||
|
||||
## Summary
|
||||
|
||||
Recover the true origin of packages whose recorded "source" is a placeholder —
|
||||
specifically the manual-upload and `rebuild.ts`-created records whose
|
||||
`sourceChannelId` points at the destination (archive) channel itself. When a
|
||||
real source channel is re-indexed and the worker encounters an archive that
|
||||
matches such a placeholder package, it backfills the real
|
||||
`sourceChannelId` / `sourceMessageId` / `sourceTopicId` / `sourceCaption` /
|
||||
`creator` (and, for records that never had one, the file listing and preview)
|
||||
onto the existing package — **without downloading the full archive**.
|
||||
|
||||
Matching is a two-stage process: a zero-download candidate lookup by
|
||||
`fileName` + total `fileSize`, confirmed by a CRC32 fingerprint read from just
|
||||
the archive's central directory via a **ranged (tail) download** of a few
|
||||
KB–MB, instead of the multi-GB whole file.
|
||||
|
||||
This is the first of four linked sub-projects. The others (creator
|
||||
normalization, provenance display, missing-files) are out of scope here and get
|
||||
their own spec → plan cycles. "Missing files" is explicitly deferred until this
|
||||
lands.
|
||||
|
||||
## Context
|
||||
|
||||
Current state (as of this design):
|
||||
|
||||
- `Package` records their origin via `sourceChannelId`, `sourceMessageId`,
|
||||
`sourceTopicId`, `sourceCaption`, and (for content dedup) `contentHash` and
|
||||
`remoteUniqueId`. `creator` is a free-text string extracted at ingestion.
|
||||
- Two ingestion paths create records with **placeholder provenance**, where the
|
||||
"source" is the destination channel, not a real origin:
|
||||
- **Manual uploads** (`worker/src/manual-upload.ts`) set
|
||||
`sourceChannelId = destChannel.id` and `sourceMessageId = destResult.messageId`.
|
||||
They *do* populate `PackageFile` (with `crc32`) from a local central-directory
|
||||
read at upload time.
|
||||
- **`rebuild.ts`** scans the destination channel and creates minimal records
|
||||
with `fileCount == 0` and **no** `PackageFile` rows.
|
||||
- The main worker upload path (`worker/src/worker.ts` → `uploadToChannel`) sends
|
||||
archives to the destination channel with **no caption**, so provenance cannot
|
||||
be recovered by re-reading destination messages — it has to come from matching
|
||||
against real source channels.
|
||||
- The scan/dedup ladder in `processOneArchiveSet` (`worker/src/worker.ts:1521`)
|
||||
is entirely **source-channel-scoped**:
|
||||
1. `findPackageByRemoteUniqueId(channel.id, …)` — same channel only
|
||||
2. `packageExistsBySourceMessage(channel.id, …)` — same channel only
|
||||
3. `findRepostedPackage(channel.id, fileName, size)` — same channel only
|
||||
4. …then full download → `packageExistsByHash(contentHash)` — global, but only
|
||||
*after* downloading the whole archive.
|
||||
- Because a placeholder package's `sourceChannelId` is the destination, checks
|
||||
1–3 never match it when a real source channel is scanned. Today the worker
|
||||
therefore downloads the entire archive, hits check 4, finds it is a duplicate,
|
||||
and skips — **wasting the download and leaving the wrong provenance in place.**
|
||||
- There is already an in-production precedent for "match an already-known
|
||||
archive during scan and enrich its metadata": when `findRepostedPackage`
|
||||
matches, the worker backfills richer *topic* context onto the existing package
|
||||
via `updatePackageTopicContext` (`worker/src/worker.ts:1608`+). Provenance
|
||||
backfill is the same move, widened from same-channel to cross-channel.
|
||||
- `remote.unique_id` is **not** a reliable cross-channel key: it identifies a
|
||||
stored file object on Telegram's servers, not the content. The same archive
|
||||
independently uploaded to two channels gets different `unique_id`s; it only
|
||||
matches for forwarded messages (same underlying file object). This is why the
|
||||
existing dedup scopes it to a single channel.
|
||||
|
||||
## Goals / Non-Goals
|
||||
|
||||
**Goals**
|
||||
|
||||
- Attribute true origin to placeholder-provenance packages during normal
|
||||
re-index scans, opportunistically (no separate pass).
|
||||
- Do it without downloading whole archives (ranged central-directory read only).
|
||||
- Be non-destructive: only ever touch packages that currently have placeholder
|
||||
provenance; never overwrite a real, non-placeholder source.
|
||||
- Be idempotent: re-running a re-index does not re-mutate or duplicate.
|
||||
|
||||
**Non-Goals (separate sub-projects / deferred)**
|
||||
|
||||
- Creator name normalization / canonical creator entity.
|
||||
- UI changes to display provenance.
|
||||
- Recovering "missing files."
|
||||
- Choosing a *preferred* origin when an archive genuinely exists in several
|
||||
source channels (first confirmed source wins).
|
||||
|
||||
## Design
|
||||
|
||||
### 1. Candidate definition
|
||||
|
||||
> A package is a backfill candidate iff **`sourceChannelId == destChannelId`
|
||||
> OR `sourceMessageId == 0`**.
|
||||
|
||||
Two placeholder shapes exist (verified against live data 2026-07-23):
|
||||
- **Manual uploads** (`manual-upload.ts`): `sourceChannelId == destChannelId`,
|
||||
real `contentHash`, real `sourceMessageId`, has a `PackageFile` listing.
|
||||
- **Rebuild records** (`rebuild.ts`): `sourceMessageId == 0n` (deliberate
|
||||
"unknown" sentinel), synthetic `contentHash = "rebuild:<destChannelId>:<destMessageId>"`,
|
||||
`fileCount == 0`, and `sourceChannelId` set to an **arbitrary fallback source
|
||||
channel** (`sourceChannels[0]`) — NOT the destination. (This is the common
|
||||
case: e.g. 59,893 records after a destination rebuild.)
|
||||
|
||||
Normal ingestion always sets a real `sourceMessageId` (> 0) and a real source
|
||||
channel, so neither marker matches a genuinely-sourced package. Both markers are
|
||||
overwritten on backfill (source channel + message become real), so a record
|
||||
stops being a candidate once fixed — this is what makes re-scans idempotent.
|
||||
|
||||
**Known limitation:** backfill does NOT rewrite a rebuild record's synthetic
|
||||
`"rebuild:"` `contentHash` (the true content hash would require a full download,
|
||||
which this feature avoids). That is acceptable — dedup after backfill relies on
|
||||
`remoteUniqueId` + name/size within the source channel, not on `contentHash`.
|
||||
|
||||
### 2. Where it hooks
|
||||
|
||||
A new step is inserted into `processOneArchiveSet` **between check #3
|
||||
(`findRepostedPackage`) and the full download**. It runs only after the
|
||||
same-channel checks have missed (so genuine same-channel reposts keep their
|
||||
existing fast paths).
|
||||
|
||||
Flow for the scanned archive set:
|
||||
|
||||
1. Stage A — **candidate lookup (zero download).** Query for a package where
|
||||
`(sourceChannelId == destChannelId OR sourceMessageId == 0)` AND
|
||||
`fileName == archiveName` AND `fileSize == totalArchiveSize`.
|
||||
(`Package` has `@@index([fileName])`.)
|
||||
- No candidate → fall through to normal ingestion unchanged.
|
||||
2. Stage B — **fingerprint confirmation (tiny download).** See §3.
|
||||
3. On confirmation → **backfill** (see §4) and return `null` (treated as a
|
||||
duplicate; no full download, no new package). Increment a `zipsBackfilled`
|
||||
counter.
|
||||
4. On mismatch / failure / ambiguity → fall through to normal ingestion (see §6).
|
||||
|
||||
### 3. Fingerprint confirmation
|
||||
|
||||
The confirmation signal is the **multiset of internal CRC32s** of the archive's
|
||||
entries (CRC32 of each entry's *uncompressed* data). This value is a property of
|
||||
the archive contents and is identical regardless of which channel hosts the
|
||||
file.
|
||||
|
||||
- **Candidate side:** for manual uploads, `PackageFile.crc32` is already
|
||||
populated from the local central-directory read — zero cost. For **rebuild
|
||||
records** (`fileCount == 0`, no `crc32`), there is nothing stored to compare
|
||||
against; obtain the candidate's fingerprint with a **second ranged tail read
|
||||
of its destination copy** (ZIP/7z — still no full download). RAR candidates
|
||||
cannot be tail-read on either side, so they fall back to name+size (see §5).
|
||||
- **Scanned-source side:** read via a **ranged (tail) download** of the central
|
||||
directory:
|
||||
- **ZIP** (incl. multipart): the End-of-Central-Directory record + central
|
||||
directory live at the tail of the last part. Download only that tail and
|
||||
parse entries. (Reuses the lightweight-listing mechanism that is
|
||||
sub-project 4's core.)
|
||||
- **7z**: a start header at byte 0 points to an end header at the tail; fetch
|
||||
both small pieces.
|
||||
- **RAR**: headers are scattered through the file — no cheap tail read. See §5.
|
||||
- **Match rule:** confirmed iff the sorted CRC32 multisets are equal **and** file
|
||||
counts are equal.
|
||||
|
||||
The comparison logic (CRC32 multiset equality) and the candidate-match predicate
|
||||
are implemented as **pure functions** with no TDLib dependency, so they are unit
|
||||
testable (see §7).
|
||||
|
||||
### 4. What gets written
|
||||
|
||||
On a confirmed match, update the candidate package in a single transaction,
|
||||
overwriting only placeholder/empty fields:
|
||||
|
||||
| Field | New value | Condition |
|
||||
|---|---|---|
|
||||
| `sourceChannelId` | scanned `channel.id` | always |
|
||||
| `sourceMessageId` | scanned `parts[0].id` | always |
|
||||
| `sourceTopicId` | `ctx.sourceTopicId` | always |
|
||||
| `sourceCaption` | scanned message caption | always |
|
||||
| `remoteUniqueId` | scanned `firstRemoteUniqueId` | always (enables future same-channel dedup via check #1) |
|
||||
| `creator` | re-derived from source (topic > filename > channel) | **always** — un-normalized; the later creator-normalization sub-project cleans it up |
|
||||
| `fileCount` + `PackageFile[]` | from the central-directory read | only if candidate had none (`fileCount == 0`) — folds in the sub-project 4 outcome for rebuild records |
|
||||
| `previewData` / `previewMsgId` | matched preview from scan | only if candidate has none |
|
||||
|
||||
**Left untouched:** `contentHash`, `destChannelId`, `destMessageId`,
|
||||
`destMessageIds` — the bytes physically live in the destination channel; that is
|
||||
correct and must not change.
|
||||
|
||||
**Transaction safety:** re-check `sourceChannelId == destChannelId` *inside* the
|
||||
transaction before writing, so a concurrent worker that already backfilled the
|
||||
record causes this one to no-op (mirrors the existing `backfill.ts` guard).
|
||||
|
||||
### 5. RAR handling
|
||||
|
||||
RAR sources cannot be tail-read, so no cheap CRC fingerprint is available.
|
||||
Decision: **backfill RAR matches on `fileName` + `fileSize` alone**, flagged as
|
||||
lower-confidence in logs and via a `SystemNotification`, so they can be audited.
|
||||
No full download.
|
||||
|
||||
This name+size-only fallback applies to any candidate that cannot produce a
|
||||
CRC32 fingerprint cheaply: a RAR **source**, a RAR **candidate**, or a rebuild
|
||||
candidate whose destination copy is RAR. ZIP/7z rebuild candidates are still
|
||||
confirmed by fingerprint via the second tail read described in §3.
|
||||
|
||||
### 6. Conflict, mismatch, ambiguity
|
||||
|
||||
- **Fingerprint mismatch** → the scanned archive is genuinely different content
|
||||
that merely shares name+size. Fall through to **normal ingestion** (download +
|
||||
index) — it is a new package for this source.
|
||||
- **Ambiguous candidates** (2+ match name+size and the fingerprint cannot
|
||||
disambiguate) → log a `SystemNotification`, backfill nothing, and fall through
|
||||
to normal ingestion; the post-download `packageExistsByHash` check still
|
||||
dedups it safely.
|
||||
- **Same archive in multiple source channels** → the **first re-indexed source
|
||||
that confirms wins.** After backfill the package has a real source, so it is no
|
||||
longer a candidate; later scans treat it as an ordinary duplicate via checks #1
|
||||
(the `remoteUniqueId` we set) or #3.
|
||||
|
||||
### 7. Idempotency
|
||||
|
||||
Falls out of the candidate definition. Once backfilled:
|
||||
- `sourceChannelId` is the real channel → no longer a candidate.
|
||||
- A re-scan of that source hits check #1 (`remoteUniqueId`, now set) or check #3
|
||||
(`findRepostedPackage`) → normal dedup skip. No re-mutation, no duplicate.
|
||||
|
||||
### 8. Error handling
|
||||
|
||||
- **Tail-download failure** (network / `FLOOD_WAIT`) → cannot confirm this round.
|
||||
Do **not** fall back to a full download and do **not** guess: leave the
|
||||
candidate untouched and let the next re-index retry. Wrap the ranged read in
|
||||
`withFloodWait` (per the TDLib skill).
|
||||
- All new TDLib calls follow the skill's patterns: `withFloodWait`, listener
|
||||
attached before the async op, client closed in `finally`.
|
||||
|
||||
### 9. Visibility
|
||||
|
||||
- Add a `zipsBackfilled` counter to `IngestionRun` activity so a re-index run
|
||||
reports "N provenance backfills" instead of silently mutating records.
|
||||
- Info log per backfill (candidate id, old vs new source, confidence:
|
||||
fingerprint | name+size-RAR).
|
||||
- `SystemNotification` for ambiguous-candidate cases.
|
||||
|
||||
## Testing
|
||||
|
||||
The repo currently has no test framework (`CLAUDE.md`: "testing is manual"). This
|
||||
work introduces a **lightweight test harness** for the worker (e.g. `vitest` or
|
||||
node's built-in `node:test`) and unit-tests the correctness-critical pure logic:
|
||||
|
||||
- **Unit (automated):**
|
||||
- CRC32 multiset fingerprint equality (equal sets, different order, differing
|
||||
counts, disjoint sets).
|
||||
- Candidate-match predicate (name+size+placeholder true/false cases).
|
||||
- Field-merge rules (which fields overwrite, which only fill-if-empty).
|
||||
- **Manual integration checklist:**
|
||||
1. Re-index a real source channel containing a known manually-uploaded pack →
|
||||
verify `sourceChannelId`/`sourceMessageId`/`sourceCaption`/`creator` are
|
||||
backfilled and no full download occurs.
|
||||
2. A rebuild-created record (`fileCount == 0`) → verify listing + provenance
|
||||
are both populated from the single tail read.
|
||||
3. A RAR pack → verify name+size backfill with the lower-confidence log/notice.
|
||||
4. Re-run the same re-index → verify it is a no-op (idempotent).
|
||||
5. A genuine name+size collision (different content) → verify it ingests as a
|
||||
new package rather than being mis-attributed.
|
||||
|
||||
## Open Questions
|
||||
|
||||
None blocking. Preferred-origin selection among multiple real sources is
|
||||
deliberately out of scope (first confirmed wins).
|
||||
|
||||
## Affected Code (indicative, for planning)
|
||||
|
||||
- `worker/src/worker.ts` — `processOneArchiveSet`: new Stage A/B step; new counter.
|
||||
- `worker/src/archive/` — ranged central-directory reader (ZIP/7z tail); shared
|
||||
with sub-project 4. Pure CRC32-fingerprint compare helper.
|
||||
- `worker/src/tdlib/download.ts` — ranged/partial download support (`offset`/`limit`).
|
||||
- `worker/src/db/queries.ts` — candidate lookup + transactional backfill update.
|
||||
- `prisma/schema.prisma` — `IngestionRun.zipsBackfilled` (and run-counter plumbing).
|
||||
- Worker test harness + first unit tests.
|
||||
@@ -0,0 +1,180 @@
|
||||
# Ranged inner-file listing for RAR & 7z — design
|
||||
|
||||
**Date:** 2026-07-27
|
||||
**Status:** Approved (design), pending spec review → implementation plan
|
||||
|
||||
## Problem
|
||||
|
||||
The reindex/provenance-backfill path (`worker/src/provenance-backfill.ts`) can index an
|
||||
archive's inner files *without* re-downloading it, by reading the file listing from a small
|
||||
ranged read of the copy already in the source/destination channel. This works **only for
|
||||
ZIP** today (ZIP keeps its central directory in a tail that `parseZipCentralDirectoryFromTail`
|
||||
reads). For **RAR and 7z**, `tryProvenanceBackfill` backfills provenance (creator, source
|
||||
channel, `remoteUniqueId`) so the file is skipped on re-scan and never re-downloaded — but it
|
||||
leaves the inner listing empty (`fileCount = 0`), because `scannedEntries` is only computed for
|
||||
ZIP.
|
||||
|
||||
Scope of the gap (rebuild placeholders with `fileCount = 0`, as of 2026-07-27):
|
||||
|
||||
| Type | Count | Total | Avg | Max | Multipart |
|
||||
|---|---|---|---|---|---|
|
||||
| 7z | 19,646 | 14 TB | 0.73 GB | 3.9 GB | 0 |
|
||||
| RAR | 12,780 | 13 TB | 1.04 GB | 116 GB | 1,016 |
|
||||
|
||||
A "just download the whole archive" fallback for all of these means ~27 TB of re-downloads —
|
||||
the exact cost this path exists to avoid.
|
||||
|
||||
## Goals
|
||||
|
||||
- Index the inner files (names + sizes; CRCs where cheaply available) of RAR and 7z
|
||||
placeholders **without** downloading the whole archive in the common case.
|
||||
- Reuse the existing, battle-tested CLI listing parsers (`parse7zOutput`,
|
||||
`parseUnrarTechnical`) rather than reimplementing filename/size/CRC extraction.
|
||||
- Keep cost proportional to **file count**, not archive size (so even the 116 GB RAR is cheap).
|
||||
- Guarantee a listing for the rare archives the cheap path can't handle, via a full-download
|
||||
fallback that respects the existing max-size guard.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- No change to ingestion of genuinely-new files (those are downloaded in full to be re-uploaded
|
||||
regardless, so a cheap listing does not help there). This feature only affects the
|
||||
provenance-backfill / skip path.
|
||||
- No new ZIP behaviour — the existing ZIP tail reader stays as-is.
|
||||
- Not attempting to list password-encrypted-header archives from ranged reads (no password);
|
||||
those take the fallback and, if oversized, are flagged.
|
||||
|
||||
## Approach (chosen: "harvest header regions → sparse file → native CLI")
|
||||
|
||||
Do the *minimum* binary parsing needed to locate an archive's header bytes, fetch only those
|
||||
via ranged reads, write them into a sparse temp file at their true offsets (data regions left
|
||||
as unwritten zero holes → ~no disk use), then run the real `7z l` / `unrar lt` and reuse the
|
||||
existing parsers. The native tools do the hard parsing (7z's LZMA-encoded headers, RAR's two
|
||||
format versions, Unicode names) — we only compute where the headers are.
|
||||
|
||||
Rejected alternatives: full native TS parsers (most custom binary code, highest risk);
|
||||
RAR5-quick-open-only (most community RARs lack it → collapses to ~13 TB of RAR downloads).
|
||||
|
||||
## Components
|
||||
|
||||
New directory `worker/src/archive/ranged/`, one focused module per concern, all returning the
|
||||
existing `FileEntry[]` type from `zip-reader.ts`:
|
||||
|
||||
- `sparse-list.ts` — `listFromSparse(parts, runner, parse) → FileEntry[] | null`, where each
|
||||
`part` is `{ fileName, size, regions: {offset, bytes}[] }`. For each part it writes a sparse
|
||||
temp file (`truncate` to `size`, then write only the header `regions` at their offsets),
|
||||
co-locates all parts in one temp dir under their real names, invokes the supplied CLI runner
|
||||
(`7z l` / `unrar lt`) on the first part, feeds stdout to the supplied `parse` fn
|
||||
(`parse7zOutput` / `parseUnrarTechnical`), and cleans up. Single-part archives are just the
|
||||
one-element case. Returns `null` on CLI error / empty parse.
|
||||
- `sevenz-ranged.ts` — `readSevenZListingRanged(client, parts) → FileEntry[] | null`.
|
||||
- `rar-ranged.ts` — `readRarListingRanged(client, parts) → FileEntry[] | null`.
|
||||
- Dispatcher in `provenance-backfill.ts`: `readScannedListingRanged(archiveType, client, parts)`
|
||||
replacing the current `if (archiveType === "ZIP")` branch; the destination-copy read in
|
||||
`resolveCandidateFingerprintEntries` gets the same dispatch.
|
||||
|
||||
`FileEntry` shape (unchanged): `{ path, fileName, extension, compressedSize, uncompressedSize, crc32 }`.
|
||||
|
||||
### 7z ranged listing
|
||||
|
||||
7z layout: 32-byte signature header at offset 0 → packed streams → end header (lists files) at
|
||||
the end; the signature header stores the end header's location.
|
||||
|
||||
1. Ranged-read `[0, 32)`; validate magic `37 7A BC AF 27 1C`. Read LE `uint64`
|
||||
`NextHeaderOffset` (byte 12) and `NextHeaderSize` (byte 20). End header is at absolute offset
|
||||
`32 + NextHeaderOffset`, length `NextHeaderSize`.
|
||||
2. Ranged-read `[32 + NextHeaderOffset, NextHeaderSize)` — the "next header".
|
||||
3. Branch on the next header's first byte (a 7z property id):
|
||||
- **`0x01` (kHeader, plain/uncompressed header):** two regions suffice —
|
||||
`{0: sigHeader}` and `{32+NextHeaderOffset: endHeader}`.
|
||||
- **`0x17` (kEncodedHeader, LZMA-compressed header):** the next header is only a *descriptor*
|
||||
whose `PackInfo` points at a packed header stream stored **in the middle** of the file (not
|
||||
at EOF). Parse the descriptor's `StreamsInfo → kPackInfo (0x06)` to read `PackPos` and the
|
||||
`PackSize`s (7z variable-length "numbers"; sum them). Ranged-read the contiguous packed
|
||||
region `[32 + PackPos, Σ PackSize)` and add it as a **third** sparse region. `7z l` then
|
||||
decodes the header from that region.
|
||||
- **anything else:** return `null` (→ fallback).
|
||||
4. `listFromSparse` with the 2 or 3 regions, runner = `7z l`. `parse7zOutput` yields
|
||||
names+sizes (`crc32: null`, as today). Return `null` on bad magic / read failure / CLI error.
|
||||
|
||||
**Why the third region is required (spike finding, 2026-07-27):** the original two-region
|
||||
(start+end) reconstruction was proven insufficient in a live test — `7z l` rejected it with
|
||||
"Cannot open the file as [7z] archive" because these archives use an *encoded* header whose
|
||||
compressed bytes live in a packed stream in the file body (a sparse hole), not at EOF. The
|
||||
`0x17` branch fetches exactly that packed region. `7z l` still never touches the file-data
|
||||
packed streams (it only lists), so those gaps stay sparse. The `read7zNumber` reader (7z's
|
||||
base-128-ish variable-length integer with a first-byte length mask) and the minimal
|
||||
`kPackInfo` walk are the only new 7z binary parsing; `7z l` still does the actual file listing.
|
||||
All 7z placeholders are single-part.
|
||||
|
||||
### RAR ranged listing
|
||||
|
||||
RAR has no index; walk the block chain, parsing only each block's **size fields** to step
|
||||
forward and harvest header bytes.
|
||||
|
||||
1. Read first ~16 bytes; detect **RAR4** (`52 61 72 21 1A 07 00`) vs **RAR5** (`…07 01 00`) and
|
||||
the signature length.
|
||||
2. From just after the signature, loop:
|
||||
- Ranged-read a header chunk (start 8 KB; if parsed `HeaderSize` exceeds it — long filenames
|
||||
— re-read exactly).
|
||||
- Minimal block-extent parse:
|
||||
- RAR5: `CRC32(4)` + vint `HeaderSize` + vint `HeaderType` + vint `HeaderFlags`; if the
|
||||
"extra area" flag (`0x0001`) → vint `ExtraAreaSize`; if the "data present" flag
|
||||
(`0x0002`) → vint `DataSize`. Next block = `pos + 4 + len(HeaderSize vint) + HeaderSize
|
||||
+ DataSize`.
|
||||
- RAR4: `HEAD_CRC(2)` + `HEAD_TYPE(1)` + `HEAD_FLAGS(2)` + `HEAD_SIZE(2)`; if flag `0x8000`
|
||||
→ `ADD_SIZE(4)`. Next block = `pos + HEAD_SIZE + ADD_SIZE`.
|
||||
- Harvest `[blockOffset, blockOffset + HeaderSize)` into the regions list.
|
||||
- Stop at the end-of-archive block or EOF.
|
||||
3. `listFromSparse` (headers present, data sparse) → `unrar lt` → `parseUnrarTechnical`. RAR
|
||||
headers carry CRC32, so RAR contributes CRCs (fingerprint disambiguation keeps working).
|
||||
|
||||
**Multipart RAR** (1,016): each volume starts with its own signature + headers. Walk **each
|
||||
part from its own signature**, reconstruct one sparse temp file per part with correct names
|
||||
(`name.part1.rar`, `.part2.rar`, …) co-located in a temp dir, and run `unrar lt` on part 1 —
|
||||
`unrar` auto-discovers co-located siblings (per the existing reader's note). The global-offset →
|
||||
`(part, offsetInPart)` mapping reuses the multipart size math the ZIP path already uses.
|
||||
|
||||
## Fallback & integration
|
||||
|
||||
- A ranged reader returning `null` = cheap read failed (bad magic, read error, walk gave up, or
|
||||
CLI error on the sparse file) → **full-download fallback**: download the whole archive, run the
|
||||
existing `readRarContents` / `read7zContents`, backfill.
|
||||
- The fallback is gated by `config.maxZipSizeMB` (the same guard used at ingest). Over the cap →
|
||||
no download; write a `SystemNotification` (`INTEGRITY_AUDIT`, WARNING) and leave the listing
|
||||
empty for manual review. This ensures nothing pathological (e.g. the 116 GB RAR) is pulled.
|
||||
- Downstream is unchanged: `compareFingerprints` already treats null/incomplete CRCs as
|
||||
"incomplete" (name-size path), and `backfillProvenance` writes entries when the candidate's
|
||||
`fileCount === 0`.
|
||||
- Observability: reuse the `zipsBackfilled` counter; add structured logs with
|
||||
`confidence: "ranged" | "full-download-fallback"` and a WARN on fallback so miss-rate is
|
||||
visible.
|
||||
|
||||
## Risks & de-risking spike (do before the full build)
|
||||
|
||||
On 3–4 real placeholder archives per format:
|
||||
|
||||
1. Confirm `downloadFileRange` returns correct bytes at **arbitrary (non-tail) offsets** —
|
||||
currently only tail-verified in production. Underpins everything; if it fails, stop and
|
||||
rethink. (Note: `range-download.ts` flags absolute-offset behaviour as pending live
|
||||
verification; tail reads are proven by the 43 ZIP backfills done 2026-07-26.)
|
||||
2. Confirm `7z l` and `unrar lt` list correctly from a **sparse reconstructed file** — single
|
||||
part first, then multipart RAR (the highest-risk case).
|
||||
|
||||
If multipart-RAR sparse reconstruction proves unreliable in the spike, multipart RAR uses the
|
||||
full-download fallback (respecting the size cap → oversized ones flagged, not downloaded).
|
||||
|
||||
## Testing
|
||||
|
||||
- **Unit (vitest, alongside `central-directory.test.ts`):** 7z signature-header parse; RAR4 &
|
||||
RAR5 block-extent walk against committed small fixtures; `sparse-list` writes the correct
|
||||
regions. Pure logic, no TDLib.
|
||||
- **Live post-deploy:** watch `zipsBackfilled` climb for RAR/7z via the ranged path; spot-check
|
||||
a handful of backfilled packages' `package_files` against a real `unrar lt` / `7z l` on a full
|
||||
download of the same file; confirm the fallback/flag path fires on a deliberately-broken case.
|
||||
|
||||
## Rollout
|
||||
|
||||
Local build + deploy (no GitHub push required), per the established recipe: build
|
||||
`worker/Dockerfile` locally, recreate the `dragonsstash-worker` container from the local image
|
||||
(no `pull`). No new DB migration. The scheduler re-runs hourly and will backfill RAR/7z
|
||||
placeholders on subsequent cycles.
|
||||
@@ -0,0 +1,211 @@
|
||||
# Forward-priority ingestion — design
|
||||
|
||||
**Date:** 2026-07-30
|
||||
**Status:** Approved (design), pending spec review → implementation plan
|
||||
|
||||
## Problem
|
||||
|
||||
The worker ingests every archive the same way regardless of whether it needs to: download the
|
||||
full file from the source channel, then re-upload the full file to the destination (archive)
|
||||
channel. That download+reupload round-trip was originally necessary because some source channels
|
||||
have "restrict saving content" (protected content) enabled, which blocks Telegram-native
|
||||
forwarding — for those channels there is no alternative to moving the bytes through the worker.
|
||||
|
||||
But most source channels do NOT restrict forwarding. For those, the round-trip is pure waste:
|
||||
Telegram can copy the message from source chat to destination chat server-side, with no bytes
|
||||
ever passing through the worker. The worker still needs to end up with the same outcome it has
|
||||
today — a destination-channel copy, a dedup-safe identity, and a full inner-file listing — just
|
||||
without paying for a download and re-upload to get there.
|
||||
|
||||
Separately, `feat/ranged-archive-listing` (merged to master ahead of this feature) already built
|
||||
exactly the missing piece: reading a ZIP/RAR/7z archive's inner-file listing via small ranged
|
||||
reads against the file wherever it currently lives (source channel, destination channel — doesn't
|
||||
matter), with no full download. It was built for backfilling listings onto already-deduped
|
||||
placeholder packages. This feature generalizes that same capability to fresh ingestion, and pairs
|
||||
it with a new native-forward upload path.
|
||||
|
||||
## Goals
|
||||
|
||||
- For channels that allow forwarding: skip download and re-upload entirely for new archives. Use
|
||||
Telegram-native forwarding from source chat to destination chat, and the existing ranged-listing
|
||||
readers to index inner files, with no full download in the common case.
|
||||
- For channels that block forwarding (or when forwarding isn't yet known): keep today's
|
||||
download+reupload pipeline exactly as-is.
|
||||
- Every ingested package — regardless of path — ends up with the same outcome as today: a
|
||||
`Package` row with a valid dedup identity, `destMessageId`/`destMessageIds`, creator, tags, and a
|
||||
full inner-file listing (`PackageFile` rows). Indexing completeness must not regress.
|
||||
- If the cheap ranged listing fails for a specific archive (bad/unsupported header, CLI error,
|
||||
etc.) in an otherwise-forwarding-eligible channel, fall back to today's full download+reupload
|
||||
pipeline for that one archive — never forward with an empty or partial listing.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- No ranged single-entry preview extraction. Forward-path packages still get a preview when a
|
||||
channel photo message matches (cheap, unrelated to archive bytes); when there's no matching
|
||||
photo, forward-path packages simply have no preview, same as any package where preview
|
||||
extraction fails today. In-archive preview extraction (unzip/unrar/7z against a local file) stays
|
||||
as a download-path-only feature. May be revisited as a follow-up if it turns out to matter.
|
||||
- No reprocessing of already-ingested packages. This only changes behavior for newly-scanned
|
||||
archives going forward.
|
||||
- No change to the bot's user-delivery leg (`bot/src/tdlib/client.ts` `copyMessageToUser`) — it
|
||||
already sends via `inputFileRemote` with no download, and is unaffected by this feature.
|
||||
- No change to `config.maxZipSizeMB` or the multipart byte-level split/repack logic. The existing
|
||||
size guard runs before either path is chosen, so nothing above the cap reaches the forward path's
|
||||
fallback-to-download step either. Splitting simply never engages on the forward path — a
|
||||
forwarded message is already within whatever size Telegram accepted when it was first uploaded.
|
||||
|
||||
## Approaches considered
|
||||
|
||||
**A — Branch inside the existing pipeline (chosen).** Add one fork point in
|
||||
`processOneArchiveSet`, immediately after the existing pre-download dedup checks: if the channel
|
||||
allows forwarding, attempt the ranged-listing + forward path; on any failure, fall through into
|
||||
today's download-based code for that one archive, unchanged. Smallest diff; reuses the existing
|
||||
dedup/retry/watermark machinery as-is; matches the file's existing forum-vs-non-forum branching
|
||||
style.
|
||||
|
||||
**B — Separate pipeline per channel.** Decide once per channel and route the whole channel through
|
||||
either a "forward module" or the existing "download module." Cleaner separation on paper, but
|
||||
duplicates the SkippedPackage/stall/watermark bookkeeping that currently lives once in
|
||||
`processArchiveSets`/`processOneArchiveSet` — higher regression risk in a large orchestration file
|
||||
with no tests at that level. Rejected.
|
||||
|
||||
**C — Strategy-object refactor.** Extract an `IngestStrategy` interface (`download` / `forward`)
|
||||
and slim `processOneArchiveSet` to delegate to it. The more "proper" abstraction, but it's a
|
||||
structural refactor of already-battle-tested code that doesn't need it for this feature to work.
|
||||
Rejected — can revisit later if a third strategy ever appears.
|
||||
|
||||
## Sequencing
|
||||
|
||||
`feat/ranged-archive-listing` merges to master first, as-is (it's complete and serves a different
|
||||
purpose already). This feature is built on a fresh branch off master afterward.
|
||||
|
||||
## Components
|
||||
|
||||
### 1. `TelegramChannel.allowsForwarding` (new column, new migration)
|
||||
|
||||
`Boolean?` — nullable, `null` means "not yet checked". Refreshed from TDLib's chat
|
||||
protected-content flag (exact field name to be confirmed against the pinned `tdl`/TDLib version
|
||||
via docs lookup during implementation — expected to be `chat.has_protected_content`) at the same
|
||||
point the worker already calls `getChat` per channel per cycle, mirroring the existing
|
||||
`isForum`/`setChannelForum` read-and-persist pattern precisely. `null` or `false` both route to the
|
||||
download path — a channel never uses the forward path on unverified permission.
|
||||
|
||||
### 2. Shared ranged-listing dispatcher
|
||||
|
||||
`readScannedListingRanged` (plus `RangedPart`, `tdlibRangeReader`, and the format-specific
|
||||
ZIP/RAR/7z readers) currently live inside `provenance-backfill.ts`. Promote the dispatcher (and
|
||||
whatever it depends on) into a shared module (e.g. `worker/src/archive/ranged/dispatch.ts`) so
|
||||
`worker.ts` can call the same no-download listing logic for fresh ingestion without a circular
|
||||
import. `provenance-backfill.ts` switches to importing from the new shared location; behavior
|
||||
unchanged for the existing backfill path.
|
||||
|
||||
### 3. `forwardArchiveToChannel` (new, `worker/src/upload/forward.ts`)
|
||||
|
||||
Mirrors `uploadToChannel`'s shape and return type (`{ messageId, messageIds }`). Uses TDLib
|
||||
`forwardMessages` to copy all parts of an archive set from the source chat to the destination chat
|
||||
in one batch call (message IDs in original order), wrapped in the same flood-wait/retry handling
|
||||
style as `uploadToChannel`. Followed by the same destination read-back verification style as
|
||||
today's post-upload check (`getMessage` on each new destination message ID, confirm a document is
|
||||
present).
|
||||
|
||||
### 4. Dedup identity for forward-path packages
|
||||
|
||||
`Package.contentHash` stays a required unique string, but forward-path packages can't hash real
|
||||
bytes. Derivation order:
|
||||
1. If the ranged listing's CRC32s are complete (ZIP/RAR today) — hash the sorted CRC32 list into a
|
||||
synthetic `fingerprint:<hash>` value, reusing `archive/fingerprint.ts`'s existing
|
||||
`crcFingerprint`.
|
||||
2. Otherwise (7z, or any incomplete-CRC case) — synthesize `forward:<remoteUniqueId>`, following
|
||||
the existing `rebuild:`-prefixed placeholder-hash precedent in `rebuild.ts`.
|
||||
|
||||
Additionally, extend repost detection: before committing to the forward path, compare the new
|
||||
listing's CRC fingerprint (via the existing `compareFingerprints`/`fingerprintsMatch` logic already
|
||||
used in `provenance-backfill.ts`'s ambiguous-candidate disambiguation) against recent Packages
|
||||
sharing the same file name + size. A fingerprint match is treated as a duplicate and skipped, same
|
||||
as today's `findRepostedPackage` handling — this is what lets a forwarded copy and a previously
|
||||
fully-downloaded copy of the same archive still dedupe against each other, despite never sharing a
|
||||
byte-hash-derived `contentHash`.
|
||||
|
||||
### 5. Fork point in `processOneArchiveSet`
|
||||
|
||||
All existing pre-download checks run first, completely unchanged, in the same order:
|
||||
`remote.unique_id` match → `packageExistsBySourceMessage` → `findRepostedPackage` (name+size) →
|
||||
cross-channel provenance backfill → size guard (`maxZipSizeMB`).
|
||||
|
||||
Then:
|
||||
|
||||
```
|
||||
if channel.allowsForwarding === true:
|
||||
entries = readScannedListingRanged(archiveType, client, scannedParts)
|
||||
if entries is not null:
|
||||
contentHash = deriveForwardContentHash(entries, remoteUniqueId)
|
||||
if fingerprintRepostCheck(entries, fileName, fileSize) finds a match:
|
||||
→ treat as duplicate, skip (same bookkeeping as today's dup path)
|
||||
destResult = forwardArchiveToChannel(client, sourceChatId, partMessageIds, destChatId)
|
||||
creator, tags ← derived from entries/filename/channel/topic, same as today
|
||||
preview ← channel-photo match only (no in-archive extraction)
|
||||
createPackageStub(...) + updatePackageWithMetadata(...), same as today
|
||||
counters.zipsForwarded++
|
||||
→ done
|
||||
else:
|
||||
→ fall through into the existing download/hash/split/upload flow below, unchanged
|
||||
(log the fallback for observability)
|
||||
else:
|
||||
→ existing download/hash/split/upload flow, completely unchanged
|
||||
```
|
||||
|
||||
### 6. Observability
|
||||
|
||||
New `zipsForwarded` counter alongside the existing `zipsFound`/`zipsDuplicate`/`zipsIngested`/
|
||||
`zipsBackfilled` counters, surfaced the same way (run activity, ingestion run summary). A WARN-level
|
||||
log line when a forwarding-eligible archive falls back to download (mirrors the existing
|
||||
`confidence: "ranged" | "full-download-fallback"` logging convention from the ranged-listing
|
||||
backfill work), so the fallback rate is visible without digging through debug logs.
|
||||
|
||||
## Data flow
|
||||
|
||||
```
|
||||
scan → pre-download dedup + size guard (unchanged)
|
||||
→ channel.allowsForwarding?
|
||||
true → ranged listing
|
||||
ok → fingerprint dedup check → forward → stub + entries + tags (no in-archive preview) → done
|
||||
null → [fall through] existing download pipeline
|
||||
false/unknown → existing download pipeline (unchanged)
|
||||
```
|
||||
|
||||
## Error handling
|
||||
|
||||
- `forwardMessages` failure (permission revoked mid-run, rate limit, transient Telegram error) —
|
||||
same `SkippedPackage`/`SystemNotification` bookkeeping as today's upload failures. Extend
|
||||
`inferSkipReason` to recognize forward-specific error text the same way it already recognizes
|
||||
upload errors.
|
||||
- Fingerprint-repost check finds multiple ambiguous same-name/size candidates that can't be
|
||||
uniquely disambiguated — same `INTEGRITY_AUDIT` notification pattern already used in
|
||||
`provenance-backfill.ts`: don't guess, surface for manual triage.
|
||||
- `allowsForwarding` unknown (channel just linked, not yet scanned by the refresh point) — treated
|
||||
as `false`; the download path runs. No channel uses an unverified forwarding permission.
|
||||
- Ranged listing throwing instead of returning `null` — treated identically to returning `null`
|
||||
(fall through to download), consistent with how the existing ranged readers already treat
|
||||
internal errors (they catch and return `null` themselves).
|
||||
|
||||
## Testing
|
||||
|
||||
- Unit tests (vitest, alongside the existing `archive/*.test.ts` and `archive/ranged/*.test.ts`
|
||||
files): the dedup-identity derivation function (fingerprint-hash vs remoteUniqueId-fallback
|
||||
branches), the extended fingerprint-based repost check, and `forwardArchiveToChannel`'s
|
||||
request-building logic against a mocked TDLib client — same style as the existing ranged-reader
|
||||
tests (pure logic, no live TDLib).
|
||||
- Live verification (manual — matches this repo's existing convention that the large
|
||||
`worker.ts`/`worker.py`-equivalent orchestration function has no automated test coverage and is
|
||||
verified live post-deploy): one forwarding-enabled test channel and one protected-content test
|
||||
channel. Confirm forward-path packages land with correct entries/tags/dedup identity and
|
||||
`destMessageIds`; confirm the protected channel still goes through the unchanged full pipeline;
|
||||
confirm a deliberately-unparseable archive in a forwarding-enabled channel correctly falls back
|
||||
to download+reupload and still ends up fully indexed.
|
||||
|
||||
## Rollout
|
||||
|
||||
Local build + deploy, following the same recipe as the ranged-archive-listing work: build
|
||||
`worker/Dockerfile` locally, recreate the `dragonsstash-worker` container from the local image (no
|
||||
`pull`, no GitHub push required). New DB migration for `TelegramChannel.allowsForwarding`. No
|
||||
changes required to the bot or app services.
|
||||
@@ -0,0 +1 @@
|
||||
ALTER TABLE "manual_upload_files" ADD COLUMN "retainedAt" TIMESTAMP(3);
|
||||
@@ -0,0 +1 @@
|
||||
ALTER TABLE "manual_upload_files" DROP COLUMN IF EXISTS "retainedAt";
|
||||
@@ -0,0 +1,5 @@
|
||||
-- AlterTable: count of packages whose provenance was backfilled during a run
|
||||
-- (opportunistic cross-channel provenance backfill). Additive, non-null with a
|
||||
-- default of 0 — no data change for existing rows.
|
||||
ALTER TABLE "ingestion_runs"
|
||||
ADD COLUMN "zipsBackfilled" INTEGER NOT NULL DEFAULT 0;
|
||||
@@ -0,0 +1,10 @@
|
||||
-- AlterTable: forward-priority ingestion support.
|
||||
-- allowsForwarding is nullable — null means "not yet checked", treated the
|
||||
-- same as false until confirmed true (safe default: download+reupload path).
|
||||
ALTER TABLE "telegram_channels"
|
||||
ADD COLUMN "allowsForwarding" BOOLEAN;
|
||||
|
||||
-- AlterTable: count of packages forwarded (no download) during a run.
|
||||
-- Additive, non-null with a default of 0 — no data change for existing rows.
|
||||
ALTER TABLE "ingestion_runs"
|
||||
ADD COLUMN "zipsForwarded" INTEGER NOT NULL DEFAULT 0;
|
||||
@@ -428,6 +428,11 @@ model TelegramChannel {
|
||||
isForum Boolean @default(false)
|
||||
isActive Boolean @default(false)
|
||||
category String? @db.VarChar(64)
|
||||
/// Whether this chat currently allows forwarding/saving (the inverse of
|
||||
/// TDLib's chat.has_protected_content). Null = not yet checked; treated the
|
||||
/// same as false everywhere in the worker (safe default: use the
|
||||
/// download+reupload path until this is confirmed true).
|
||||
allowsForwarding Boolean?
|
||||
createdAt DateTime @default(now())
|
||||
updatedAt DateTime @updatedAt
|
||||
|
||||
@@ -569,6 +574,8 @@ model IngestionRun {
|
||||
zipsFound Int @default(0)
|
||||
zipsDuplicate Int @default(0)
|
||||
zipsIngested Int @default(0)
|
||||
zipsBackfilled Int @default(0)
|
||||
zipsForwarded Int @default(0)
|
||||
errorMessage String?
|
||||
|
||||
// Live activity tracking — written by worker in real-time
|
||||
|
||||
@@ -1,7 +1,23 @@
|
||||
import { redirect } from "next/navigation";
|
||||
import { auth } from "@/lib/auth";
|
||||
import { prisma } from "@/lib/prisma";
|
||||
import { Sidebar } from "@/components/layout/sidebar";
|
||||
import { Header } from "@/components/layout/header";
|
||||
|
||||
export default function AppLayout({ children }: { children: React.ReactNode }) {
|
||||
export default async function AppLayout({ children }: { children: React.ReactNode }) {
|
||||
// Guard against a stale JWT session whose user no longer exists in the
|
||||
// database (e.g. after a DB reset). The signed cookie still passes edge
|
||||
// middleware, but every downstream query keyed on session.user.id would fail.
|
||||
// Send such sessions to /logout, which clears the cookie and returns to login.
|
||||
const session = await auth();
|
||||
if (session?.user?.id) {
|
||||
const user = await prisma.user.findUnique({
|
||||
where: { id: session.user.id },
|
||||
select: { id: true },
|
||||
});
|
||||
if (!user) redirect("/logout");
|
||||
}
|
||||
|
||||
return (
|
||||
<div className="flex h-screen overflow-hidden">
|
||||
<div className="hidden lg:block">
|
||||
|
||||
@@ -500,14 +500,9 @@ export function StlTable({
|
||||
</Badge>
|
||||
)}
|
||||
</TabsTrigger>
|
||||
<TabsTrigger value="ungrouped" className="gap-1.5">
|
||||
Ungrouped
|
||||
{ungroupedTotalCount > 0 && (
|
||||
<Badge variant="secondary" className="h-5 px-1.5 text-[10px]">
|
||||
{ungroupedTotalCount}
|
||||
</Badge>
|
||||
)}
|
||||
</TabsTrigger>
|
||||
{/* "Ungrouped" tab hidden: the STL list is now flat and grouping is
|
||||
no longer surfaced here. The tab content below is kept (unreachable)
|
||||
to avoid churn; remove it and its data fetch if grouping is dropped. */}
|
||||
</TabsList>
|
||||
|
||||
<TabsContent value="packages" className="space-y-4">
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
import { signOut } from "@/lib/auth";
|
||||
|
||||
// Server-side sign-out that clears the JWT session cookie and redirects to the
|
||||
// login page. Used to recover from a stale session whose user no longer exists
|
||||
// in the database (e.g. after a DB reset), which a client-only signOut can't
|
||||
// reach because the app crashes before rendering the user menu.
|
||||
export async function GET() {
|
||||
await signOut({ redirectTo: "/login" });
|
||||
}
|
||||
@@ -1,20 +1,37 @@
|
||||
import { Prisma } from "@prisma/client";
|
||||
import { prisma } from "@/lib/prisma";
|
||||
|
||||
const DEFAULT_SETTINGS = {
|
||||
lowStockThreshold: 20,
|
||||
currency: "EUR",
|
||||
theme: "dark",
|
||||
units: "metric",
|
||||
} as const;
|
||||
|
||||
export async function getUserSettings(userId: string) {
|
||||
let settings = await prisma.userSettings.findUnique({
|
||||
where: { userId },
|
||||
});
|
||||
|
||||
if (!settings) {
|
||||
try {
|
||||
settings = await prisma.userSettings.create({
|
||||
data: {
|
||||
userId,
|
||||
lowStockThreshold: 20,
|
||||
currency: "EUR",
|
||||
theme: "dark",
|
||||
units: "metric",
|
||||
},
|
||||
data: { userId, ...DEFAULT_SETTINGS },
|
||||
});
|
||||
} catch (err) {
|
||||
// The session's user may no longer exist (e.g. a stale JWT cookie after a
|
||||
// database reset). Creating settings then hits a foreign-key violation
|
||||
// (P2003). Don't crash the Server Component render — return unsaved
|
||||
// defaults. The (app) layout guard redirects such stale sessions to
|
||||
// sign-out, so this fallback is only ever momentarily visible.
|
||||
if (
|
||||
err instanceof Prisma.PrismaClientKnownRequestError &&
|
||||
err.code === "P2003"
|
||||
) {
|
||||
return { id: "", userId, ...DEFAULT_SETTINGS };
|
||||
}
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
return settings;
|
||||
|
||||
+12
-12
@@ -135,31 +135,31 @@ export async function listDisplayItems(options: {
|
||||
const sortCol = sortBy === "fileName" ? `"fileName"` : sortBy === "fileSize" ? `"fileSize"` : `"indexedAt"`;
|
||||
const sortDir = order === "asc" ? "ASC" : "DESC";
|
||||
|
||||
// Step 1: Count display items
|
||||
// NOTE: The STL list is intentionally FLAT — every package is its own display
|
||||
// row regardless of packageGroupId. Grouping is no longer surfaced in this
|
||||
// view (the creator column/filter organizes the list instead). PackageGroup
|
||||
// rows and the manual grouping actions still exist in the DB/UI; they just
|
||||
// don't drive this list's layout anymore.
|
||||
|
||||
// Step 1: Count display items (one per package)
|
||||
const countResult = await prisma.$queryRawUnsafe<[{ count: bigint }]>(
|
||||
`SELECT COUNT(*) AS count FROM (
|
||||
SELECT DISTINCT COALESCE(p."packageGroupId", p."id") AS display_id
|
||||
FROM packages p
|
||||
${whereClause}
|
||||
) AS display_items`,
|
||||
`SELECT COUNT(*) AS count FROM packages p ${whereClause}`,
|
||||
...params
|
||||
);
|
||||
const total = Number(countResult[0].count);
|
||||
|
||||
// Step 2: Get display item IDs for this page
|
||||
// Step 2: Get package IDs for this page
|
||||
const limitParam = paramIdx++;
|
||||
const offsetParam = paramIdx++;
|
||||
const displayRows = await prisma.$queryRawUnsafe<
|
||||
{ display_id: string; display_type: string }[]
|
||||
>(
|
||||
`SELECT
|
||||
COALESCE(p."packageGroupId", p."id") AS display_id,
|
||||
CASE WHEN p."packageGroupId" IS NOT NULL THEN 'group' ELSE 'package' END AS display_type,
|
||||
MAX(p.${sortCol}) AS sort_value
|
||||
p."id" AS display_id,
|
||||
'package' AS display_type,
|
||||
p.${sortCol} AS sort_value
|
||||
FROM packages p
|
||||
${whereClause}
|
||||
GROUP BY COALESCE(p."packageGroupId", p."id"),
|
||||
CASE WHEN p."packageGroupId" IS NOT NULL THEN 'group' ELSE 'package' END
|
||||
ORDER BY sort_value ${sortDir}
|
||||
LIMIT $${limitParam} OFFSET $${offsetParam}`,
|
||||
...params, limit, (page - 1) * limit
|
||||
|
||||
Generated
+1034
-1
File diff suppressed because it is too large
Load Diff
+5
-2
@@ -6,7 +6,9 @@
|
||||
"scripts": {
|
||||
"build": "tsc",
|
||||
"start": "node dist/index.js",
|
||||
"dev": "tsx watch src/index.ts"
|
||||
"dev": "tsx watch src/index.ts",
|
||||
"test": "vitest run",
|
||||
"test:watch": "vitest"
|
||||
},
|
||||
"dependencies": {
|
||||
"@prisma/adapter-pg": "^7.4.0",
|
||||
@@ -23,6 +25,7 @@
|
||||
"@types/yauzl": "^2.10.3",
|
||||
"prisma": "^7.4.0",
|
||||
"tsx": "^4.21.0",
|
||||
"typescript": "^5"
|
||||
"typescript": "^5",
|
||||
"vitest": "^3.2.4"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,69 @@
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { parseZipCentralDirectoryFromTail } from "./central-directory.js";
|
||||
import { crc32 } from "zlib"; // Node 20+ exposes zlib.crc32
|
||||
|
||||
// Build a minimal STORE (no compression) ZIP in-memory with the given files.
|
||||
function buildStoreZip(files: { name: string; data: Buffer }[]): Buffer {
|
||||
const chunks: Buffer[] = [];
|
||||
const central: Buffer[] = [];
|
||||
let offset = 0;
|
||||
for (const f of files) {
|
||||
const crc = crc32(f.data) >>> 0;
|
||||
const nameBuf = Buffer.from(f.name, "utf8");
|
||||
const local = Buffer.alloc(30);
|
||||
local.writeUInt32LE(0x04034b50, 0);
|
||||
local.writeUInt16LE(20, 4); // version needed
|
||||
local.writeUInt16LE(0, 6); // flags
|
||||
local.writeUInt16LE(0, 8); // method = store
|
||||
local.writeUInt32LE(crc, 14);
|
||||
local.writeUInt32LE(f.data.length, 18); // compressed
|
||||
local.writeUInt32LE(f.data.length, 22); // uncompressed
|
||||
local.writeUInt16LE(nameBuf.length, 26);
|
||||
local.writeUInt16LE(0, 28); // extra len
|
||||
const localHeader = Buffer.concat([local, nameBuf, f.data]);
|
||||
chunks.push(localHeader);
|
||||
|
||||
const cd = Buffer.alloc(46);
|
||||
cd.writeUInt32LE(0x02014b50, 0);
|
||||
cd.writeUInt16LE(20, 4); cd.writeUInt16LE(20, 6);
|
||||
cd.writeUInt16LE(0, 8); cd.writeUInt16LE(0, 10);
|
||||
cd.writeUInt32LE(crc, 16);
|
||||
cd.writeUInt32LE(f.data.length, 20);
|
||||
cd.writeUInt32LE(f.data.length, 24);
|
||||
cd.writeUInt16LE(nameBuf.length, 28);
|
||||
cd.writeUInt32LE(offset, 42); // local header offset
|
||||
central.push(Buffer.concat([cd, nameBuf]));
|
||||
offset += localHeader.length;
|
||||
}
|
||||
const cdBuf = Buffer.concat(central);
|
||||
const cdOffset = offset;
|
||||
const eocd = Buffer.alloc(22);
|
||||
eocd.writeUInt32LE(0x06054b50, 0);
|
||||
eocd.writeUInt16LE(files.length, 8);
|
||||
eocd.writeUInt16LE(files.length, 10);
|
||||
eocd.writeUInt32LE(cdBuf.length, 12);
|
||||
eocd.writeUInt32LE(cdOffset, 16);
|
||||
return Buffer.concat([...chunks, cdBuf, eocd]);
|
||||
}
|
||||
|
||||
describe("parseZipCentralDirectoryFromTail", () => {
|
||||
it("lists entries with correct names, sizes, and crc32", () => {
|
||||
const zip = buildStoreZip([
|
||||
{ name: "models/dragon.stl", data: Buffer.from("DRAGON") },
|
||||
{ name: "readme.txt", data: Buffer.from("hello world") },
|
||||
]);
|
||||
const entries = parseZipCentralDirectoryFromTail(zip, 0);
|
||||
expect(entries.map((e) => e.fileName).sort()).toEqual(["dragon.stl", "readme.txt"]);
|
||||
const dragon = entries.find((e) => e.fileName === "dragon.stl")!;
|
||||
expect(dragon.path).toBe("models/dragon.stl");
|
||||
expect(dragon.uncompressedSize).toBe(6n);
|
||||
expect(dragon.crc32).toMatch(/^[0-9a-f]{8}$/);
|
||||
});
|
||||
|
||||
it("throws when the central directory begins before the tail window", () => {
|
||||
const zip = buildStoreZip([{ name: "a.txt", data: Buffer.alloc(100) }]);
|
||||
// Provide only the last 30 bytes but claim they start at offset (len-30):
|
||||
const tail = zip.subarray(zip.length - 30);
|
||||
expect(() => parseZipCentralDirectoryFromTail(tail, zip.length - 30)).toThrow(RangeError);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,90 @@
|
||||
import path from "path";
|
||||
import type { FileEntry } from "./zip-reader.js";
|
||||
|
||||
export const MIN_ZIP_TAIL_BYTES = 65_557;
|
||||
|
||||
const EOCD_SIG = 0x06054b50;
|
||||
const CD_SIG = 0x02014b50;
|
||||
|
||||
function extOf(name: string): string | null {
|
||||
const e = path.extname(name).replace(/^\./, "").toLowerCase();
|
||||
return e === "" ? null : e;
|
||||
}
|
||||
|
||||
/** Parse a ZIP central directory from the tail of an archive. */
|
||||
export function parseZipCentralDirectoryFromTail(tail: Buffer, tailStart: number): FileEntry[] {
|
||||
// 1. Find EOCD by scanning backward for its signature.
|
||||
let eocd = -1;
|
||||
for (let i = tail.length - 22; i >= 0; i--) {
|
||||
if (tail.readUInt32LE(i) === EOCD_SIG) { eocd = i; break; }
|
||||
}
|
||||
if (eocd < 0) throw new RangeError("EOCD not found in tail");
|
||||
|
||||
let cdSize = tail.readUInt32LE(eocd + 12);
|
||||
let cdOffset = tail.readUInt32LE(eocd + 16);
|
||||
|
||||
// ZIP64: sizes/offsets of 0xFFFFFFFF mean "see ZIP64 EOCD".
|
||||
if (cdOffset === 0xffffffff || cdSize === 0xffffffff) {
|
||||
const locSig = 0x07064b50;
|
||||
let loc = -1;
|
||||
for (let i = eocd - 20; i >= 0; i--) {
|
||||
if (tail.readUInt32LE(i) === locSig) { loc = i; break; }
|
||||
}
|
||||
if (loc < 0) throw new RangeError("ZIP64 EOCD locator not in tail");
|
||||
const z64Abs = Number(tail.readBigUInt64LE(loc + 8)); // absolute offset of ZIP64 EOCD
|
||||
const z64 = z64Abs - tailStart;
|
||||
if (z64 < 0) throw new RangeError("ZIP64 EOCD before tail window");
|
||||
cdSize = Number(tail.readBigUInt64LE(z64 + 40));
|
||||
cdOffset = Number(tail.readBigUInt64LE(z64 + 48));
|
||||
}
|
||||
|
||||
// 2. Map the absolute central-directory offset into the tail buffer.
|
||||
const cdLocal = cdOffset - tailStart;
|
||||
if (cdLocal < 0 || cdLocal + cdSize > tail.length) {
|
||||
throw new RangeError("Central directory begins before tail window");
|
||||
}
|
||||
|
||||
// 3. Walk central-directory headers.
|
||||
const entries: FileEntry[] = [];
|
||||
let p = cdLocal;
|
||||
const end = cdLocal + cdSize;
|
||||
while (p + 46 <= end && tail.readUInt32LE(p) === CD_SIG) {
|
||||
let crc = tail.readUInt32LE(p + 16) >>> 0;
|
||||
let comp = BigInt(tail.readUInt32LE(p + 20));
|
||||
let uncomp = BigInt(tail.readUInt32LE(p + 24));
|
||||
const nameLen = tail.readUInt16LE(p + 28);
|
||||
const extraLen = tail.readUInt16LE(p + 30);
|
||||
const commentLen = tail.readUInt16LE(p + 32);
|
||||
const name = tail.toString("utf8", p + 46, p + 46 + nameLen);
|
||||
|
||||
// ZIP64 extra field overrides 0xFFFFFFFF sizes.
|
||||
if (comp === 0xffffffffn || uncomp === 0xffffffffn) {
|
||||
let ep = p + 46 + nameLen;
|
||||
const extraEnd = ep + extraLen;
|
||||
while (ep + 4 <= extraEnd) {
|
||||
const id = tail.readUInt16LE(ep);
|
||||
const sz = tail.readUInt16LE(ep + 2);
|
||||
if (id === 0x0001) {
|
||||
let fp = ep + 4;
|
||||
if (uncomp === 0xffffffffn) { uncomp = tail.readBigUInt64LE(fp); fp += 8; }
|
||||
if (comp === 0xffffffffn) { comp = tail.readBigUInt64LE(fp); fp += 8; }
|
||||
}
|
||||
ep += 4 + sz;
|
||||
}
|
||||
}
|
||||
|
||||
const isDir = name.endsWith("/");
|
||||
if (!isDir) {
|
||||
entries.push({
|
||||
path: name,
|
||||
fileName: path.basename(name),
|
||||
extension: extOf(name),
|
||||
compressedSize: comp,
|
||||
uncompressedSize: uncomp,
|
||||
crc32: crc !== 0 ? crc.toString(16).padStart(8, "0") : null,
|
||||
});
|
||||
}
|
||||
p += 46 + nameLen + extraLen + commentLen;
|
||||
}
|
||||
return entries;
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { crcFingerprint, fingerprintsMatch } from "./fingerprint.js";
|
||||
import type { FileEntry } from "./zip-reader.js";
|
||||
|
||||
const fe = (crc: string | null): FileEntry => ({
|
||||
path: "a", fileName: "a", extension: null,
|
||||
compressedSize: 0n, uncompressedSize: 0n, crc32: crc,
|
||||
});
|
||||
|
||||
describe("crcFingerprint", () => {
|
||||
it("sorts crcs and marks complete", () => {
|
||||
expect(crcFingerprint([fe("00ff"), fe("00aa")])).toEqual({ crcs: ["00aa", "00ff"], complete: true });
|
||||
});
|
||||
it("is incomplete when any crc is null", () => {
|
||||
expect(crcFingerprint([fe("00aa"), fe(null)]).complete).toBe(false);
|
||||
});
|
||||
it("is incomplete when empty", () => {
|
||||
expect(crcFingerprint([]).complete).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("fingerprintsMatch", () => {
|
||||
it("matches identical crc multisets regardless of order", () => {
|
||||
expect(fingerprintsMatch([fe("01"), fe("02")], [fe("02"), fe("01")])).toBe(true);
|
||||
});
|
||||
it("rejects different counts", () => {
|
||||
expect(fingerprintsMatch([fe("01")], [fe("01"), fe("02")])).toBe(false);
|
||||
});
|
||||
it("rejects disjoint sets", () => {
|
||||
expect(fingerprintsMatch([fe("01")], [fe("09")])).toBe(false);
|
||||
});
|
||||
it("rejects when either side is incomplete", () => {
|
||||
expect(fingerprintsMatch([fe("01"), fe(null)], [fe("01"), fe("02")])).toBe(false);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,21 @@
|
||||
import type { FileEntry } from "./zip-reader.js";
|
||||
|
||||
export function crcFingerprint(entries: FileEntry[]): { crcs: string[]; complete: boolean } {
|
||||
if (entries.length === 0) return { crcs: [], complete: false };
|
||||
const crcs: string[] = [];
|
||||
let complete = true;
|
||||
for (const e of entries) {
|
||||
if (e.crc32 == null) { complete = false; continue; }
|
||||
crcs.push(e.crc32.toLowerCase());
|
||||
}
|
||||
crcs.sort();
|
||||
return { crcs, complete };
|
||||
}
|
||||
|
||||
export function fingerprintsMatch(a: FileEntry[], b: FileEntry[]): boolean {
|
||||
const fa = crcFingerprint(a);
|
||||
const fb = crcFingerprint(b);
|
||||
if (!fa.complete || !fb.complete) return false;
|
||||
if (fa.crcs.length !== fb.crcs.length) return false;
|
||||
return fa.crcs.every((c, i) => c === fb.crcs[i]);
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { createHash } from "crypto";
|
||||
import { deriveForwardContentHash } from "./forward-identity.js";
|
||||
import type { FileEntry } from "./zip-reader.js";
|
||||
|
||||
function entry(crc32: string | null): FileEntry {
|
||||
return { path: "a", fileName: "a", extension: null, compressedSize: 1n, uncompressedSize: 1n, crc32 };
|
||||
}
|
||||
|
||||
describe("deriveForwardContentHash", () => {
|
||||
it("hashes the sorted CRC list when all entries have a CRC32 (ZIP/RAR)", () => {
|
||||
const entries = [entry("BBBB"), entry("AAAA")];
|
||||
const expectedHash = createHash("sha256").update(["aaaa", "bbbb"].join(",")).digest("hex");
|
||||
expect(deriveForwardContentHash(entries, "unique-1", "chan-1", 42n)).toBe(`fingerprint:${expectedHash}`);
|
||||
});
|
||||
|
||||
it("falls back to remoteUniqueId when CRCs are incomplete (7z today)", () => {
|
||||
const entries = [entry(null), entry("AAAA")];
|
||||
expect(deriveForwardContentHash(entries, "unique-42", "chan-1", 42n)).toBe("forward:unique-42");
|
||||
});
|
||||
|
||||
it("falls back to sourceChannelId+sourceMessageId when there's no CRC and no remoteUniqueId", () => {
|
||||
const entries = [entry(null)];
|
||||
expect(deriveForwardContentHash(entries, null, "chan-1", 42n)).toBe("forward:chan-1:42");
|
||||
});
|
||||
|
||||
it("falls back past an empty entries list the same way", () => {
|
||||
expect(deriveForwardContentHash([], null, "chan-1", 7n)).toBe("forward:chan-1:7");
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,33 @@
|
||||
import { createHash } from "crypto";
|
||||
import { crcFingerprint } from "./fingerprint.js";
|
||||
import type { FileEntry } from "./zip-reader.js";
|
||||
|
||||
/**
|
||||
* Derive a Package.contentHash-compatible identity string for a forward-path
|
||||
* package (no downloaded bytes exist to hash directly). Priority order:
|
||||
* 1. A CRC32-fingerprint hash, when the ranged listing's CRCs are complete
|
||||
* (ZIP/RAR today) — the strongest available signal, since it lets
|
||||
* forward-path and download-path copies of the same archive still
|
||||
* collide/dedupe on identical content.
|
||||
* 2. TDLib's remote.unique_id, when CRCs are incomplete (7z today has none).
|
||||
* 3. sourceChannelId+sourceMessageId, as a last-resort unique value so the
|
||||
* required-unique Package.contentHash column is always satisfiable.
|
||||
* Follows the same `<prefix>:<value>` synthetic-hash convention already used
|
||||
* by `rebuild.ts`'s `rebuild:${destChannelId}:${destMessageId}` placeholder.
|
||||
*/
|
||||
export function deriveForwardContentHash(
|
||||
entries: FileEntry[],
|
||||
remoteUniqueId: string | null,
|
||||
sourceChannelId: string,
|
||||
sourceMessageId: bigint,
|
||||
): string {
|
||||
const fp = crcFingerprint(entries);
|
||||
if (fp.complete && fp.crcs.length > 0) {
|
||||
const hash = createHash("sha256").update(fp.crcs.join(",")).digest("hex");
|
||||
return `fingerprint:${hash}`;
|
||||
}
|
||||
if (remoteUniqueId) {
|
||||
return `forward:${remoteUniqueId}`;
|
||||
}
|
||||
return `forward:${sourceChannelId}:${sourceMessageId}`;
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
import { describe, it, expect, vi } from "vitest";
|
||||
|
||||
const candidate = {
|
||||
id: "pkg-1", archiveType: "ZIP", fileName: "a.zip", fileCount: 3, fileSize: 100n,
|
||||
destMessageId: 1n, destMessageIds: [1n], destChannel: { telegramId: 999n },
|
||||
};
|
||||
|
||||
vi.mock("../db/queries.js", () => ({
|
||||
findFingerprintDedupCandidates: vi.fn(async () => [candidate]),
|
||||
}));
|
||||
const resolveMock = vi.fn(async (..._args: unknown[]) => [{ path: "x", fileName: "x", extension: null, compressedSize: 1n, uncompressedSize: 1n, crc32: "AAAA" }]);
|
||||
const compareMock = vi.fn();
|
||||
vi.mock("../provenance-backfill.js", () => ({
|
||||
resolveCandidateFingerprintEntries: (...args: unknown[]) => resolveMock(...args),
|
||||
compareFingerprints: (...args: unknown[]) => compareMock(...args),
|
||||
}));
|
||||
|
||||
import { checkFingerprintRepost } from "./forward-repost-check.js";
|
||||
import type { FileEntry } from "./zip-reader.js";
|
||||
|
||||
const newEntries: FileEntry[] = [{ path: "x", fileName: "x", extension: null, compressedSize: 1n, uncompressedSize: 1n, crc32: "AAAA" }];
|
||||
|
||||
describe("checkFingerprintRepost", () => {
|
||||
it("reports a duplicate when a candidate's fingerprint matches", async () => {
|
||||
compareMock.mockReturnValueOnce("match");
|
||||
const result = await checkFingerprintRepost({} as never, newEntries, "a.zip", 100n);
|
||||
expect(result).toEqual({ isDuplicate: true, matchedPackageId: "pkg-1" });
|
||||
});
|
||||
|
||||
it("reports no duplicate when no candidate matches", async () => {
|
||||
compareMock.mockReturnValueOnce("mismatch");
|
||||
const result = await checkFingerprintRepost({} as never, newEntries, "a.zip", 100n);
|
||||
expect(result).toEqual({ isDuplicate: false, matchedPackageId: null });
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,33 @@
|
||||
import type { Client } from "tdl";
|
||||
import type { FileEntry } from "./zip-reader.js";
|
||||
import { compareFingerprints, resolveCandidateFingerprintEntries } from "../provenance-backfill.js";
|
||||
import { findFingerprintDedupCandidates } from "../db/queries.js";
|
||||
|
||||
export interface FingerprintRepostResult {
|
||||
isDuplicate: boolean;
|
||||
matchedPackageId: string | null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Cross-channel duplicate check for the forward-priority path: compare the
|
||||
* new archive's CRC fingerprint against every existing Package sharing its
|
||||
* name+size, regardless of which channel or ingestion path produced them.
|
||||
* This is what lets a forwarded copy dedupe against a previously
|
||||
* fully-downloaded copy of the same archive, despite never sharing a
|
||||
* byte-hash-derived contentHash.
|
||||
*/
|
||||
export async function checkFingerprintRepost(
|
||||
client: Client,
|
||||
entries: FileEntry[],
|
||||
fileName: string,
|
||||
fileSize: bigint,
|
||||
): Promise<FingerprintRepostResult> {
|
||||
const candidates = await findFingerprintDedupCandidates(fileName, fileSize);
|
||||
for (const candidate of candidates) {
|
||||
const candidateEntries = await resolveCandidateFingerprintEntries(client, candidate);
|
||||
if (compareFingerprints(entries, candidateEntries) === "match") {
|
||||
return { isDuplicate: true, matchedPackageId: candidate.id };
|
||||
}
|
||||
}
|
||||
return { isDuplicate: false, matchedPackageId: null };
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { readScannedListingRanged } from "./dispatch.js";
|
||||
|
||||
describe("readScannedListingRanged", () => {
|
||||
it("returns null for an unknown archive type without calling the reader", async () => {
|
||||
const read = async () => Buffer.alloc(0);
|
||||
const result = await readScannedListingRanged(
|
||||
"DOCUMENT",
|
||||
{ invoke: async () => ({}) } as never,
|
||||
[{ fileId: "1", fileSize: 100n, fileName: "a.pdf" }],
|
||||
);
|
||||
expect(result).toBeNull();
|
||||
void read; // unused placeholder kept out of the dispatch call — DOCUMENT never reaches a reader
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,62 @@
|
||||
import type { Client } from "tdl";
|
||||
import { downloadFileRange } from "../../tdlib/range-download.js";
|
||||
import { parseZipCentralDirectoryFromTail, MIN_ZIP_TAIL_BYTES } from "../central-directory.js";
|
||||
import { childLogger } from "../../util/logger.js";
|
||||
import type { FileEntry } from "../zip-reader.js";
|
||||
import { readSevenZListingRanged, type RangedPart } from "./sevenz-ranged.js";
|
||||
import { readRarListingRanged } from "./rar-ranged.js";
|
||||
import { tdlibRangeReader } from "./range-reader.js";
|
||||
|
||||
const log = childLogger("ranged-dispatch");
|
||||
|
||||
/**
|
||||
* Read a ZIP central directory from the tail of a (possibly multipart)
|
||||
* archive. `parts` is ordered; only the LAST part carries the EOCD record.
|
||||
* `fileSize` on each part is that part's own size (NOT the whole-archive
|
||||
* total) so the download offset stays within that part's bounds, while
|
||||
* `tailStart` passed to the parser is the logical whole-archive offset
|
||||
* (preceding parts' sizes + the offset within the last part).
|
||||
*/
|
||||
export async function readScannedZipListing(
|
||||
client: Client,
|
||||
parts: { fileId: string; fileSize: bigint }[],
|
||||
): Promise<FileEntry[] | null> {
|
||||
if (parts.length === 0) return null;
|
||||
const lastPart = parts[parts.length - 1];
|
||||
const precedingSize = parts.slice(0, -1).reduce((sum, p) => sum + Number(p.fileSize), 0);
|
||||
const lastSize = Number(lastPart.fileSize);
|
||||
for (const tailBytes of [MIN_ZIP_TAIL_BYTES, MIN_ZIP_TAIL_BYTES * 4]) {
|
||||
const partOffset = Math.max(0, lastSize - tailBytes);
|
||||
const downloadLen = Math.min(tailBytes, lastSize);
|
||||
try {
|
||||
const buf = await downloadFileRange(client, lastPart.fileId, partOffset, downloadLen, lastPart.fileSize);
|
||||
const tailStart = precedingSize + partOffset;
|
||||
return parseZipCentralDirectoryFromTail(buf, tailStart);
|
||||
} catch (err) {
|
||||
if (err instanceof RangeError) continue; // try a larger tail
|
||||
log.warn({ err, fileId: lastPart.fileId }, "ranged ZIP listing failed");
|
||||
return null;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Dispatch a (no-download) inner-file listing read by archive type. Used both
|
||||
* by the provenance-backfill path (reading an already-uploaded copy) and the
|
||||
* forward-priority ingestion path (reading the source channel's copy before
|
||||
* any download/forward decision is made) — the read itself only needs
|
||||
* {fileId, fileSize, fileName}, so it doesn't matter which channel the file
|
||||
* currently lives in.
|
||||
*/
|
||||
export async function readScannedListingRanged(
|
||||
archiveType: string,
|
||||
client: Client,
|
||||
parts: RangedPart[],
|
||||
): Promise<FileEntry[] | null> {
|
||||
const read = tdlibRangeReader(client);
|
||||
if (archiveType === "ZIP") return readScannedZipListing(client, parts);
|
||||
if (archiveType === "SEVEN_Z") return readSevenZListingRanged(parts, read);
|
||||
if (archiveType === "RAR") return readRarListingRanged(parts, read);
|
||||
return null;
|
||||
}
|
||||
@@ -0,0 +1,27 @@
|
||||
import { describe, it, expect, vi } from "vitest";
|
||||
|
||||
// logLevel is required here too (not just maxZipSizeMB/tempDir) because this
|
||||
// mock replaces the config module for the whole test-file graph, including
|
||||
// util/logger.ts's module-level `pino({ level: config.logLevel })` call —
|
||||
// pino throws at import time if level is undefined.
|
||||
vi.mock("../../util/config.js", () => ({ config: { maxZipSizeMB: 1, tempDir: "/tmp", logLevel: "info" } }));
|
||||
const created: unknown[] = [];
|
||||
vi.mock("../../db/client.js", () => ({
|
||||
db: { systemNotification: { create: async (a: unknown) => { created.push(a); } } },
|
||||
}));
|
||||
|
||||
import { fullDownloadListing } from "./fallback.js";
|
||||
|
||||
describe("fullDownloadListing", () => {
|
||||
it("refuses to download over the size cap and records a notification", async () => {
|
||||
const res = await fullDownloadListing({
|
||||
client: {} as never,
|
||||
parts: [{ fileId: "1", fileSize: 2n * 1024n * 1024n * 1024n, fileName: "big.rar" }],
|
||||
archiveType: "RAR",
|
||||
totalSize: 2n * 1024n * 1024n * 1024n,
|
||||
fileName: "big.rar",
|
||||
});
|
||||
expect(res).toBeNull();
|
||||
expect(created).toHaveLength(1);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,55 @@
|
||||
import { mkdtemp, rm } from "fs/promises";
|
||||
import path from "path";
|
||||
import type { Client } from "tdl";
|
||||
import { config } from "../../util/config.js";
|
||||
import { db } from "../../db/client.js";
|
||||
import { childLogger } from "../../util/logger.js";
|
||||
import { downloadFile } from "../../tdlib/download.js";
|
||||
import { read7zContents } from "../sevenz-reader.js";
|
||||
import { readRarContents } from "../rar-reader.js";
|
||||
import type { FileEntry } from "../zip-reader.js";
|
||||
import type { RangedPart } from "./sevenz-ranged.js";
|
||||
|
||||
const log = childLogger("ranged-fallback");
|
||||
|
||||
export async function fullDownloadListing(args: {
|
||||
client: Client;
|
||||
parts: RangedPart[];
|
||||
archiveType: string;
|
||||
totalSize: bigint;
|
||||
fileName: string;
|
||||
}): Promise<FileEntry[] | null> {
|
||||
const capBytes = BigInt(config.maxZipSizeMB) * 1024n * 1024n;
|
||||
if (args.totalSize > capBytes) {
|
||||
await db.systemNotification.create({
|
||||
data: {
|
||||
type: "INTEGRITY_AUDIT",
|
||||
severity: "WARNING",
|
||||
title: `Listing skipped (over size cap): ${args.fileName}`,
|
||||
message: `Ranged listing failed and the archive (${args.totalSize} bytes) exceeds WORKER_MAX_ZIP_SIZE_MB; not downloaded. Inner files left unindexed.`,
|
||||
context: { fileName: args.fileName, archiveType: args.archiveType },
|
||||
},
|
||||
});
|
||||
log.warn({ fileName: args.fileName }, "fallback skipped — over size cap");
|
||||
return null;
|
||||
}
|
||||
const dir = await mkdtemp(path.join(config.tempDir, "fallback-"));
|
||||
const paths: string[] = [];
|
||||
try {
|
||||
for (const p of args.parts) {
|
||||
const dest = path.join(dir, p.fileName);
|
||||
await downloadFile(args.client, p.fileId, dest, p.fileSize, p.fileName, () => {});
|
||||
paths.push(dest);
|
||||
}
|
||||
const entries =
|
||||
args.archiveType === "SEVEN_Z" ? await read7zContents(paths[0])
|
||||
: args.archiveType === "RAR" ? await readRarContents(paths[0])
|
||||
: [];
|
||||
return entries.length > 0 ? entries : null;
|
||||
} catch (err) {
|
||||
log.warn({ err, fileName: args.fileName }, "full-download fallback failed");
|
||||
return null;
|
||||
} finally {
|
||||
await rm(dir, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
import type { Client } from "tdl";
|
||||
import { downloadFileRange } from "../../tdlib/range-download.js";
|
||||
|
||||
export type RangeReader = (
|
||||
fileId: string,
|
||||
offset: number,
|
||||
length: number,
|
||||
partSize: bigint,
|
||||
) => Promise<Buffer>;
|
||||
|
||||
export function tdlibRangeReader(client: Client): RangeReader {
|
||||
return (fileId, offset, length, partSize) =>
|
||||
downloadFileRange(client, fileId, offset, length, partSize);
|
||||
}
|
||||
@@ -0,0 +1,150 @@
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { readVint, detectRarSignature, parseRar5BlockExtent, parseRar4BlockExtent, walkRarVolume, readRarListingRanged } from "./rar-ranged.js";
|
||||
import type { RangeReader } from "./range-reader.js";
|
||||
|
||||
describe("readVint", () => {
|
||||
it("reads single-byte and multi-byte values (base-128 LE)", () => {
|
||||
expect(readVint(Buffer.from([0x08]), 0)).toEqual({ value: 8, bytes: 1 });
|
||||
// 0x80,0x01 => 0 | (1<<7) = 128
|
||||
expect(readVint(Buffer.from([0x80, 0x01]), 0)).toEqual({ value: 128, bytes: 2 });
|
||||
});
|
||||
});
|
||||
|
||||
describe("detectRarSignature", () => {
|
||||
it("detects RAR5 and RAR4", () => {
|
||||
expect(detectRarSignature(Buffer.from([0x52,0x61,0x72,0x21,0x1a,0x07,0x01,0x00]))).toEqual({ version: 5, sigLen: 8 });
|
||||
expect(detectRarSignature(Buffer.from([0x52,0x61,0x72,0x21,0x1a,0x07,0x00]))).toEqual({ version: 4, sigLen: 7 });
|
||||
expect(detectRarSignature(Buffer.alloc(8))).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe("parseRar5BlockExtent", () => {
|
||||
it("computes header+data extent and flags end-of-archive", () => {
|
||||
// CRC32(4) | HeaderSize vint=5 | Type vint=2 (file) | Flags vint=2 (data present) | DataSize vint=100 | (pad to headerSize)
|
||||
const b = Buffer.concat([
|
||||
Buffer.from([0,0,0,0]), // CRC
|
||||
Buffer.from([0x05]), // HeaderSize = 5 (bytes after this vint)
|
||||
Buffer.from([0x02]), // Type = 2 (file)
|
||||
Buffer.from([0x02]), // Flags = 0x02 -> data present
|
||||
Buffer.from([0x64]), // DataSize = 100
|
||||
Buffer.from([0x00, 0x00]), // padding to fill HeaderSize(5): Type+Flags+DataSize=3, +2 pad =5
|
||||
]);
|
||||
const ext = parseRar5BlockExtent(b, 0);
|
||||
// headerBytes = 4 (CRC) + 1 (HeaderSize vint) + 5 (HeaderSize) = 10
|
||||
expect(ext.headerBytes).toBe(10);
|
||||
expect(ext.dataSize).toBe(100);
|
||||
expect(ext.isEnd).toBe(false);
|
||||
|
||||
const endBlk = Buffer.from([0,0,0,0, 0x02, 0x05, 0x00]); // HeaderSize=2, Type=5(end), Flags=0
|
||||
const e2 = parseRar5BlockExtent(endBlk, 0);
|
||||
expect(e2.isEnd).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe("parseRar4BlockExtent", () => {
|
||||
it("computes extent with ADD_SIZE when flag 0x8000 is set", () => {
|
||||
// CRC(2) TYPE(1)=0x74 FLAGS(2)=0x8000 HEAD_SIZE(2)=11 ADD_SIZE(4)=200
|
||||
const b = Buffer.alloc(11);
|
||||
b.writeUInt8(0x74, 2);
|
||||
b.writeUInt16LE(0x8000, 3);
|
||||
b.writeUInt16LE(11, 5);
|
||||
b.writeUInt32LE(200, 7);
|
||||
const ext = parseRar4BlockExtent(b, 0);
|
||||
expect(ext.headerBytes).toBe(11);
|
||||
expect(ext.dataSize).toBe(200);
|
||||
expect(ext.isEnd).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
// Build a synthetic RAR5 volume: signature + main header + 2 file blocks (each
|
||||
// with data) + end block. We only need extents to be walkable.
|
||||
function buildRar5Volume(): Buffer {
|
||||
const sig = Buffer.from([0x52,0x61,0x72,0x21,0x1a,0x07,0x01,0x00]);
|
||||
const block = (type: number, flags: number, dataSize: number, pad = 0) => {
|
||||
const body = [Buffer.from([type]), Buffer.from([flags])];
|
||||
if (flags & 0x0002) body.push(Buffer.from([dataSize])); // DataSize (<=127 for test)
|
||||
if (pad) body.push(Buffer.alloc(pad));
|
||||
const bodyBuf = Buffer.concat(body);
|
||||
const hs = Buffer.from([bodyBuf.length]); // HeaderSize vint (<=127)
|
||||
const header = Buffer.concat([Buffer.alloc(4), hs, bodyBuf]); // CRC(4)+HeaderSize+body
|
||||
const data = Buffer.alloc(flags & 0x0002 ? dataSize : 0, 0xEE);
|
||||
return Buffer.concat([header, data]);
|
||||
};
|
||||
const main = block(1, 0, 0); // main archive header, no data
|
||||
const f1 = block(2, 0x02, 20); // file header + 20 bytes data
|
||||
const f2 = block(2, 0x02, 30); // file header + 30 bytes data
|
||||
const end = block(5, 0, 0); // end of archive
|
||||
return Buffer.concat([sig, main, f1, f2, end]);
|
||||
}
|
||||
|
||||
describe("walkRarVolume", () => {
|
||||
it("harvests every block header and stops at end-of-archive", async () => {
|
||||
const vol = buildRar5Volume();
|
||||
const read: RangeReader = async (_id, offset, length) => vol.subarray(offset, offset + length);
|
||||
const regions = await walkRarVolume(read, { fileId: "1", fileSize: BigInt(vol.length), fileName: "a.rar" }, 5, 8);
|
||||
expect(regions).not.toBeNull();
|
||||
// main + 2 files + end = 4 header regions
|
||||
expect(regions!).toHaveLength(4);
|
||||
// First region starts right after the 8-byte signature
|
||||
expect(regions![0].offset).toBe(8);
|
||||
});
|
||||
|
||||
it("returns null when a block claims an absurd header size (corrupt/desynced)", async () => {
|
||||
// RAR5 block with HeaderSize vint encoding a value > 8MB.
|
||||
// Encode 9_000_000 as RAR vint: bytes little-endian 7-bit groups with continuation bit.
|
||||
function encodeVint(n: number): number[] {
|
||||
const out: number[] = [];
|
||||
while (n >= 0x80) {
|
||||
out.push((n & 0x7f) | 0x80);
|
||||
n = Math.floor(n / 128);
|
||||
}
|
||||
out.push(n);
|
||||
return out;
|
||||
}
|
||||
const sig = Buffer.from([0x52, 0x61, 0x72, 0x21, 0x1a, 0x07, 0x01, 0x00]); // RAR5 signature
|
||||
const hsVint = encodeVint(9_000_000);
|
||||
// Block = CRC(4) + HeaderSize vint(9MB) + Type(1 byte) + Flags(1 byte)
|
||||
const block = Buffer.concat([Buffer.alloc(4), Buffer.from(hsVint), Buffer.from([0x02, 0x00])]);
|
||||
const vol = Buffer.concat([sig, block]);
|
||||
const size = 20 * 1024 * 1024;
|
||||
const read = async (_id: string, offset: number, length: number) => {
|
||||
if (offset >= vol.length) return Buffer.alloc(0);
|
||||
return vol.subarray(offset, Math.min(offset + length, vol.length));
|
||||
};
|
||||
const regions = await walkRarVolume(read, { fileId: "1", fileSize: BigInt(size), fileName: "c.rar" }, 5, 8);
|
||||
expect(regions).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe("readRarListingRanged (single part)", () => {
|
||||
it("returns null cleanly when the reconstructed file isn't a real RAR", async () => {
|
||||
const vol = buildRar5Volume();
|
||||
const read: RangeReader = async (_id, offset, length) => vol.subarray(offset, offset + length);
|
||||
const res = await readRarListingRanged([{ fileId: "1", fileSize: BigInt(vol.length), fileName: "a.rar" }], read);
|
||||
expect(res === null || Array.isArray(res)).toBe(true); // real unrar parse covered live
|
||||
});
|
||||
});
|
||||
|
||||
describe("readRarListingRanged (multipart)", () => {
|
||||
it("walks each volume from its own signature and reconstructs all parts", async () => {
|
||||
const vol = buildRar5Volume(); // reuse from Task 6 test
|
||||
// Two volumes with identical structure; each RangeReader read is scoped by fileId.
|
||||
const byId: Record<string, Buffer> = { p1: vol, p2: vol };
|
||||
const reads: Record<string, number> = { p1: 0, p2: 0 };
|
||||
const read: RangeReader = async (fileId, offset, length) => {
|
||||
reads[fileId]++;
|
||||
return byId[fileId].subarray(offset, offset + length);
|
||||
};
|
||||
const res = await readRarListingRanged(
|
||||
[
|
||||
{ fileId: "p1", fileSize: BigInt(vol.length), fileName: "x.part1.rar" },
|
||||
{ fileId: "p2", fileSize: BigInt(vol.length), fileName: "x.part2.rar" },
|
||||
],
|
||||
read,
|
||||
);
|
||||
// Both volumes were walked (each read at least its signature + blocks).
|
||||
expect(reads.p1).toBeGreaterThan(0);
|
||||
expect(reads.p2).toBeGreaterThan(0);
|
||||
expect(res === null || Array.isArray(res)).toBe(true);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,110 @@
|
||||
const RAR4_SIG = Buffer.from([0x52, 0x61, 0x72, 0x21, 0x1a, 0x07, 0x00]);
|
||||
const RAR5_SIG = Buffer.from([0x52, 0x61, 0x72, 0x21, 0x1a, 0x07, 0x01, 0x00]);
|
||||
|
||||
export function readVint(buf: Buffer, pos: number): { value: number; bytes: number } {
|
||||
let value = 0, shift = 0, bytes = 0;
|
||||
while (pos + bytes < buf.length) {
|
||||
const b = buf[pos + bytes];
|
||||
value += (b & 0x7f) * Math.pow(2, shift); // Math.pow keeps >32-bit sizes exact up to 2^53
|
||||
bytes++;
|
||||
if ((b & 0x80) === 0) return { value, bytes };
|
||||
shift += 7;
|
||||
if (shift > 63) break;
|
||||
}
|
||||
throw new RangeError("incomplete RAR vint");
|
||||
}
|
||||
|
||||
export function detectRarSignature(buf: Buffer): { version: 4 | 5; sigLen: number } | null {
|
||||
if (buf.length >= 8 && buf.subarray(0, 8).equals(RAR5_SIG)) return { version: 5, sigLen: 8 };
|
||||
if (buf.length >= 7 && buf.subarray(0, 7).equals(RAR4_SIG)) return { version: 4, sigLen: 7 };
|
||||
return null;
|
||||
}
|
||||
|
||||
export interface BlockExtent { headerBytes: number; dataSize: number; isEnd: boolean }
|
||||
|
||||
// RAR5: CRC32(4) | HeaderSize(vint) | HeaderType(vint) | HeaderFlags(vint)
|
||||
// [ExtraAreaSize(vint) if flags&0x0001] [DataSize(vint) if flags&0x0002] ...
|
||||
export function parseRar5BlockExtent(buf: Buffer, pos: number): BlockExtent {
|
||||
let p = pos + 4; // skip CRC32
|
||||
const hs = readVint(buf, p); p += hs.bytes;
|
||||
const headerBytes = 4 + hs.bytes + hs.value; // CRC + HeaderSize-vint + HeaderSize
|
||||
const type = readVint(buf, p); p += type.bytes;
|
||||
const flags = readVint(buf, p); p += flags.bytes;
|
||||
if (flags.value & 0x0001) { const ea = readVint(buf, p); p += ea.bytes; } // extra area size (skip)
|
||||
let dataSize = 0;
|
||||
if (flags.value & 0x0002) { const ds = readVint(buf, p); p += ds.bytes; dataSize = ds.value; }
|
||||
return { headerBytes, dataSize, isEnd: type.value === 5 };
|
||||
}
|
||||
|
||||
// RAR4: HEAD_CRC(2) | HEAD_TYPE(1) | HEAD_FLAGS(2) | HEAD_SIZE(2) [ADD_SIZE(4) if flags&0x8000]
|
||||
export function parseRar4BlockExtent(buf: Buffer, pos: number): BlockExtent {
|
||||
const type = buf.readUInt8(pos + 2);
|
||||
const flags = buf.readUInt16LE(pos + 3);
|
||||
const headSize = buf.readUInt16LE(pos + 5);
|
||||
const dataSize = (flags & 0x8000) ? buf.readUInt32LE(pos + 7) : 0;
|
||||
return { headerBytes: headSize, dataSize, isEnd: type === 0x7b };
|
||||
}
|
||||
|
||||
import type { FileEntry } from "../zip-reader.js";
|
||||
import type { RangeReader } from "./range-reader.js";
|
||||
import type { RangedPart } from "./sevenz-ranged.js";
|
||||
import { listFromSparse, type SparsePart } from "./sparse-list.js";
|
||||
import { readRarContents } from "../rar-reader.js";
|
||||
import { childLogger } from "../../util/logger.js";
|
||||
|
||||
const rlog = childLogger("rar-ranged");
|
||||
const MAX_RAR_BLOCKS = 50000;
|
||||
const MAX_RAR_HEADER_BYTES = 8 * 1024 * 1024; // 8 MB — real RAR block headers are far smaller; guards against a corrupt/desynced HeaderSize
|
||||
const HEADER_CHUNK = 8192;
|
||||
|
||||
export async function walkRarVolume(
|
||||
read: RangeReader,
|
||||
part: RangedPart,
|
||||
version: 4 | 5,
|
||||
sigLen: number,
|
||||
): Promise<{ offset: number; bytes: Buffer }[] | null> {
|
||||
const size = Number(part.fileSize);
|
||||
const regions: { offset: number; bytes: Buffer }[] = [];
|
||||
let pos = sigLen;
|
||||
let blocks = 0;
|
||||
try {
|
||||
while (pos < size) {
|
||||
if (++blocks > MAX_RAR_BLOCKS) return null;
|
||||
const chunkLen = Math.min(HEADER_CHUNK, size - pos);
|
||||
let chunk = await read(part.fileId, pos, chunkLen, part.fileSize);
|
||||
const ext = version === 5 ? parseRar5BlockExtent(chunk, 0) : parseRar4BlockExtent(chunk, 0);
|
||||
if (ext.headerBytes > MAX_RAR_HEADER_BYTES) return null;
|
||||
// Ensure we have the full header bytes to harvest (long filenames).
|
||||
let headerBuf = chunk;
|
||||
if (ext.headerBytes > chunk.length) {
|
||||
headerBuf = await read(part.fileId, pos, Math.min(ext.headerBytes, size - pos), part.fileSize);
|
||||
}
|
||||
regions.push({ offset: pos, bytes: headerBuf.subarray(0, Math.min(ext.headerBytes, size - pos)) });
|
||||
if (ext.isEnd) break;
|
||||
const advance = ext.headerBytes + ext.dataSize;
|
||||
if (advance <= 0) return null;
|
||||
if (pos + advance > size) break; // data clamped at the volume boundary (multipart continuation)
|
||||
pos += advance;
|
||||
}
|
||||
return regions;
|
||||
} catch (err) {
|
||||
rlog.warn({ err, fileId: part.fileId }, "RAR volume walk failed");
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
export async function readRarListingRanged(
|
||||
parts: RangedPart[],
|
||||
read: RangeReader,
|
||||
): Promise<FileEntry[] | null> {
|
||||
const sparseParts: SparsePart[] = [];
|
||||
for (const part of parts) {
|
||||
const head = await read(part.fileId, 0, 16, part.fileSize);
|
||||
const sig = detectRarSignature(head);
|
||||
if (!sig) return null;
|
||||
const regions = await walkRarVolume(read, part, sig.version, sig.sigLen);
|
||||
if (!regions) return null;
|
||||
sparseParts.push({ fileName: part.fileName, size: Number(part.fileSize), regions });
|
||||
}
|
||||
return listFromSparse(sparseParts, readRarContents);
|
||||
}
|
||||
@@ -0,0 +1,133 @@
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { parseSevenZSignatureHeader } from "./sevenz-ranged.js";
|
||||
|
||||
const MAGIC = Buffer.from([0x37, 0x7a, 0xbc, 0xaf, 0x27, 0x1c]);
|
||||
|
||||
function buildSignatureHeader(nextOffset: bigint, nextSize: bigint): Buffer {
|
||||
const buf = Buffer.alloc(32);
|
||||
MAGIC.copy(buf, 0);
|
||||
buf.writeUInt8(0, 6); buf.writeUInt8(4, 7); // version 0.4
|
||||
buf.writeUInt32LE(0, 8); // StartHeaderCRC (unused here)
|
||||
buf.writeBigUInt64LE(nextOffset, 12);
|
||||
buf.writeBigUInt64LE(nextSize, 20);
|
||||
buf.writeUInt32LE(0, 28); // NextHeaderCRC (unused here)
|
||||
return buf;
|
||||
}
|
||||
|
||||
describe("parseSevenZSignatureHeader", () => {
|
||||
it("reads NextHeaderOffset and NextHeaderSize", () => {
|
||||
const buf = buildSignatureHeader(1_000_000n, 4096n);
|
||||
expect(parseSevenZSignatureHeader(buf)).toEqual({ nextHeaderOffset: 1_000_000, nextHeaderSize: 4096 });
|
||||
});
|
||||
|
||||
it("returns null on bad magic", () => {
|
||||
expect(parseSevenZSignatureHeader(Buffer.alloc(32))).toBeNull();
|
||||
});
|
||||
|
||||
it("returns null when shorter than 32 bytes", () => {
|
||||
expect(parseSevenZSignatureHeader(MAGIC)).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
import { readSevenZListingRanged } from "./sevenz-ranged.js";
|
||||
import type { RangeReader } from "./range-reader.js";
|
||||
|
||||
describe("readSevenZListingRanged", () => {
|
||||
it("reads the signature + end-header regions and reconstructs for 7z l", async () => {
|
||||
const size = 5_000_000;
|
||||
const endHeaderOffset = 4_900_000; // absolute
|
||||
const nextHeaderOffset = endHeaderOffset - 32;
|
||||
const sig = Buffer.alloc(32);
|
||||
Buffer.from([0x37, 0x7a, 0xbc, 0xaf, 0x27, 0x1c]).copy(sig, 0);
|
||||
sig.writeBigUInt64LE(BigInt(nextHeaderOffset), 12);
|
||||
sig.writeBigUInt64LE(100n, 20);
|
||||
|
||||
const reads: { offset: number; length: number }[] = [];
|
||||
const read: RangeReader = async (_id, offset, length) => {
|
||||
reads.push({ offset, length });
|
||||
if (offset === 0) return sig.subarray(0, length);
|
||||
return Buffer.alloc(length, 0xAB); // stand-in end-header bytes
|
||||
};
|
||||
|
||||
// Inject a fake lister via the module boundary: readSevenZListingRanged
|
||||
// calls listFromSparse(parts, read7zContents). We assert the ranged reads
|
||||
// it issued; the sparse file + real 7z is covered by live verification.
|
||||
const entries = await readSevenZListingRanged(
|
||||
[{ fileId: "1", fileSize: BigInt(size), fileName: "a.7z" }],
|
||||
read,
|
||||
);
|
||||
// entries may be null here because the stand-in bytes aren't a real 7z;
|
||||
// the contract under test is the ranged-read offsets:
|
||||
expect(reads[0]).toEqual({ offset: 0, length: 32 });
|
||||
expect(reads[1]).toEqual({ offset: endHeaderOffset, length: 100 });
|
||||
expect(entries === null || Array.isArray(entries)).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
import { read7zNumber, locate7zEncodedHeaderPack } from "./sevenz-ranged.js";
|
||||
|
||||
describe("read7zNumber", () => {
|
||||
it("reads a single-byte number", () => {
|
||||
expect(read7zNumber(Buffer.from([0x2a]), 0)).toEqual({ value: 42, next: 1 });
|
||||
});
|
||||
it("reads a two-byte number (first-byte length mask + LE trailing byte)", () => {
|
||||
// 1000 = 0x03E8: low byte 0xE8, high nibble 0x03 -> first = 0x80|0x03 = 0x83, trailing 0xE8
|
||||
expect(read7zNumber(Buffer.from([0x83, 0xe8]), 0)).toEqual({ value: 1000, next: 2 });
|
||||
// 500 = 0x01F4 -> first 0x81, trailing 0xF4
|
||||
expect(read7zNumber(Buffer.from([0x81, 0xf4]), 0)).toEqual({ value: 500, next: 2 });
|
||||
});
|
||||
it("throws when pos starts past the buffer end (short read)", () => {
|
||||
expect(() => read7zNumber(Buffer.from([0x2a]), 5)).toThrow(RangeError);
|
||||
expect(() => read7zNumber(Buffer.alloc(0), 0)).toThrow(RangeError);
|
||||
});
|
||||
});
|
||||
|
||||
describe("locate7zEncodedHeaderPack", () => {
|
||||
it("parses PackPos and summed PackSize from an encoded header", () => {
|
||||
// kEncodedHeader, kPackInfo, PackPos=1000([0x83,0xe8]), NumStreams=1([0x01]),
|
||||
// kSize, PackSize=500([0x81,0xf4])
|
||||
const enc = Buffer.from([0x17, 0x06, 0x83, 0xe8, 0x01, 0x09, 0x81, 0xf4]);
|
||||
expect(locate7zEncodedHeaderPack(enc)).toEqual({ packPos: 1000, packSize: 500 });
|
||||
});
|
||||
it("returns null for a plain (kHeader 0x01) header", () => {
|
||||
expect(locate7zEncodedHeaderPack(Buffer.from([0x01, 0x04]))).toBeNull();
|
||||
});
|
||||
it("sums multiple pack streams", () => {
|
||||
// PackPos=0([0x00]), NumStreams=2([0x02]), kSize, sizes 10([0x0a]) + 20([0x14])
|
||||
const enc = Buffer.from([0x17, 0x06, 0x00, 0x02, 0x09, 0x0a, 0x14]);
|
||||
expect(locate7zEncodedHeaderPack(enc)).toEqual({ packPos: 0, packSize: 30 });
|
||||
});
|
||||
});
|
||||
|
||||
describe("readSevenZListingRanged (encoded header)", () => {
|
||||
it("fetches the mid-file packed-header region as a third read", async () => {
|
||||
const size = 5_000_000;
|
||||
const nextHeaderOffset = 4_000_000; // relative to end of 32-byte sig header
|
||||
const endStart = 32 + nextHeaderOffset; // absolute
|
||||
const packPos = 1000; // relative to end of sig header
|
||||
const packStart = 32 + packPos; // absolute
|
||||
const packSize = 500;
|
||||
const sig = Buffer.alloc(32);
|
||||
Buffer.from([0x37, 0x7a, 0xbc, 0xaf, 0x27, 0x1c]).copy(sig, 0);
|
||||
sig.writeBigUInt64LE(BigInt(nextHeaderOffset), 12);
|
||||
sig.writeBigUInt64LE(8n, 20); // NextHeaderSize = 8 (the encoded-header descriptor below)
|
||||
const encHeader = Buffer.from([0x17, 0x06, 0x83, 0xe8, 0x01, 0x09, 0x81, 0xf4]);
|
||||
|
||||
const reads: { offset: number; length: number }[] = [];
|
||||
const read = async (_id: string, offset: number, length: number) => {
|
||||
reads.push({ offset, length });
|
||||
if (offset === 0) return sig.subarray(0, length);
|
||||
if (offset === endStart) return encHeader.subarray(0, length);
|
||||
return Buffer.alloc(length, 0xcd); // stand-in packed-header bytes
|
||||
};
|
||||
|
||||
const entries = await readSevenZListingRanged(
|
||||
[{ fileId: "1", fileSize: BigInt(size), fileName: "a.7z" }],
|
||||
read,
|
||||
);
|
||||
expect(reads[0]).toEqual({ offset: 0, length: 32 });
|
||||
expect(reads[1]).toEqual({ offset: endStart, length: 8 });
|
||||
expect(reads[2]).toEqual({ offset: packStart, length: packSize });
|
||||
expect(entries === null || Array.isArray(entries)).toBe(true);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,112 @@
|
||||
import type { FileEntry } from "../zip-reader.js";
|
||||
import { read7zContents } from "../sevenz-reader.js";
|
||||
import { listFromSparse } from "./sparse-list.js";
|
||||
import type { RangeReader } from "./range-reader.js";
|
||||
import { childLogger } from "../../util/logger.js";
|
||||
|
||||
const log = childLogger("sevenz-ranged");
|
||||
|
||||
const SEVENZ_MAGIC = Buffer.from([0x37, 0x7a, 0xbc, 0xaf, 0x27, 0x1c]);
|
||||
|
||||
const K_HEADER = 0x01;
|
||||
const K_ENCODED_HEADER = 0x17;
|
||||
const K_PACK_INFO = 0x06;
|
||||
const K_SIZE = 0x09;
|
||||
|
||||
/** Read a 7z variable-length number: first byte is a length mask, followed by
|
||||
* little-endian bytes. Math.pow keeps values exact above 2^31. */
|
||||
export function read7zNumber(buf: Buffer, pos: number): { value: number; next: number } {
|
||||
if (pos >= buf.length) throw new RangeError("7z number reads past buffer end");
|
||||
const first = buf[pos];
|
||||
let mask = 0x80;
|
||||
let value = 0;
|
||||
let p = pos + 1;
|
||||
for (let i = 0; i < 8; i++) {
|
||||
if ((first & mask) === 0) {
|
||||
value += (first & (mask - 1)) * Math.pow(2, 8 * i);
|
||||
return { value, next: p };
|
||||
}
|
||||
if (p >= buf.length) throw new RangeError("7z number overruns buffer");
|
||||
value += buf[p] * Math.pow(2, 8 * i);
|
||||
p++;
|
||||
mask >>= 1;
|
||||
}
|
||||
return { value, next: p };
|
||||
}
|
||||
|
||||
/** For an encoded (kEncodedHeader) 7z next-header, return the absolute-ish
|
||||
* location of the packed header stream(s): PackPos (relative to end of the
|
||||
* 32-byte signature header) and the summed PackSize. Null if not encoded or
|
||||
* the StreamsInfo doesn't start with PackInfo as expected. */
|
||||
export function locate7zEncodedHeaderPack(
|
||||
nextHeader: Buffer,
|
||||
): { packPos: number; packSize: number } | null {
|
||||
let p = 0;
|
||||
if (nextHeader[p] !== K_ENCODED_HEADER) return null;
|
||||
p++;
|
||||
if (nextHeader[p] !== K_PACK_INFO) return null;
|
||||
p++;
|
||||
const packPos = read7zNumber(nextHeader, p); p = packPos.next;
|
||||
const numStreams = read7zNumber(nextHeader, p); p = numStreams.next;
|
||||
if (nextHeader[p] !== K_SIZE) return null;
|
||||
p++;
|
||||
let total = 0;
|
||||
for (let i = 0; i < numStreams.value; i++) {
|
||||
const s = read7zNumber(nextHeader, p); p = s.next;
|
||||
total += s.value;
|
||||
}
|
||||
return { packPos: packPos.value, packSize: total };
|
||||
}
|
||||
|
||||
export function parseSevenZSignatureHeader(
|
||||
buf: Buffer,
|
||||
): { nextHeaderOffset: number; nextHeaderSize: number } | null {
|
||||
if (buf.length < 32) return null;
|
||||
if (!buf.subarray(0, 6).equals(SEVENZ_MAGIC)) return null;
|
||||
return {
|
||||
nextHeaderOffset: Number(buf.readBigUInt64LE(12)),
|
||||
nextHeaderSize: Number(buf.readBigUInt64LE(20)),
|
||||
};
|
||||
}
|
||||
|
||||
export interface RangedPart { fileId: string; fileSize: bigint; fileName: string }
|
||||
|
||||
export async function readSevenZListingRanged(
|
||||
parts: RangedPart[],
|
||||
read: RangeReader,
|
||||
): Promise<FileEntry[] | null> {
|
||||
const part = parts[0];
|
||||
if (!part) return null;
|
||||
const size = Number(part.fileSize);
|
||||
try {
|
||||
const sig = await read(part.fileId, 0, 32, part.fileSize);
|
||||
const parsed = parseSevenZSignatureHeader(sig);
|
||||
if (!parsed) return null;
|
||||
const endStart = 32 + parsed.nextHeaderOffset;
|
||||
if (endStart < 0 || endStart + parsed.nextHeaderSize > size) return null;
|
||||
const endHeader = await read(part.fileId, endStart, parsed.nextHeaderSize, part.fileSize);
|
||||
|
||||
const regions = [
|
||||
{ offset: 0, bytes: sig },
|
||||
{ offset: endStart, bytes: endHeader },
|
||||
];
|
||||
|
||||
const headerType = endHeader[0];
|
||||
if (headerType === K_ENCODED_HEADER) {
|
||||
// Compressed header: its packed bytes live mid-file, not at EOF. Fetch them.
|
||||
const pack = locate7zEncodedHeaderPack(endHeader);
|
||||
if (!pack) return null;
|
||||
const packStart = 32 + pack.packPos;
|
||||
if (packStart < 0 || packStart + pack.packSize > size) return null;
|
||||
const packBytes = await read(part.fileId, packStart, pack.packSize, part.fileSize);
|
||||
regions.push({ offset: packStart, bytes: packBytes });
|
||||
} else if (headerType !== K_HEADER) {
|
||||
return null; // unknown next-header type
|
||||
}
|
||||
|
||||
return listFromSparse([{ fileName: part.fileName, size, regions }], read7zContents);
|
||||
} catch (err) {
|
||||
log.warn({ err, fileId: part.fileId }, "ranged 7z listing failed");
|
||||
return null;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,41 @@
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { open } from "fs/promises";
|
||||
import { listFromSparse } from "./sparse-list.js";
|
||||
|
||||
describe("listFromSparse", () => {
|
||||
it("writes each region at its offset into a sparse file and passes the path to the lister", async () => {
|
||||
const size = 1_000_000;
|
||||
const regions = [
|
||||
{ offset: 0, bytes: Buffer.from("HEAD") },
|
||||
{ offset: size - 4, bytes: Buffer.from("TAIL") },
|
||||
];
|
||||
let seenPath = "";
|
||||
const entries = await listFromSparse(
|
||||
[{ fileName: "sample.7z", size, regions }],
|
||||
async (firstPartPath) => {
|
||||
seenPath = firstPartPath;
|
||||
const fh = await open(firstPartPath, "r");
|
||||
try {
|
||||
const head = Buffer.alloc(4); await fh.read(head, 0, 4, 0);
|
||||
const tail = Buffer.alloc(4); await fh.read(tail, 0, 4, size - 4);
|
||||
const hole = Buffer.alloc(4); await fh.read(hole, 0, 4, 500_000);
|
||||
expect(head.toString()).toBe("HEAD");
|
||||
expect(tail.toString()).toBe("TAIL");
|
||||
expect(hole.equals(Buffer.alloc(4))).toBe(true); // gap is zero
|
||||
} finally { await fh.close(); }
|
||||
return [{ path: "a/b.stl", fileName: "b.stl", extension: "stl", compressedSize: 1n, uncompressedSize: 1n, crc32: null }];
|
||||
},
|
||||
);
|
||||
expect(seenPath.endsWith("sample.7z")).toBe(true);
|
||||
expect(entries).not.toBeNull();
|
||||
expect(entries!).toHaveLength(1);
|
||||
});
|
||||
|
||||
it("returns null when the lister yields no entries", async () => {
|
||||
const res = await listFromSparse(
|
||||
[{ fileName: "x.7z", size: 100, regions: [{ offset: 0, bytes: Buffer.from("A") }] }],
|
||||
async () => [],
|
||||
);
|
||||
expect(res).toBeNull();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,52 @@
|
||||
import { mkdtemp, open, rm } from "fs/promises";
|
||||
import path from "path";
|
||||
import { config } from "../../util/config.js";
|
||||
import { childLogger } from "../../util/logger.js";
|
||||
import type { FileEntry } from "../zip-reader.js";
|
||||
|
||||
const log = childLogger("sparse-list");
|
||||
|
||||
export interface SparsePart {
|
||||
fileName: string;
|
||||
size: number;
|
||||
regions: { offset: number; bytes: Buffer }[];
|
||||
}
|
||||
|
||||
export type SparseLister = (firstPartPath: string) => Promise<FileEntry[]>;
|
||||
|
||||
/**
|
||||
* Reconstruct archive header bytes into sparse temp files (data areas left as
|
||||
* zero holes), run `lister` on the first part, return its entries.
|
||||
* Returns null on any error or when the lister finds nothing.
|
||||
*/
|
||||
export async function listFromSparse(
|
||||
parts: SparsePart[],
|
||||
lister: SparseLister,
|
||||
): Promise<FileEntry[] | null> {
|
||||
if (parts.length === 0) return null;
|
||||
const dir = await mkdtemp(path.join(config.tempDir, "ranged-"));
|
||||
try {
|
||||
let firstPath = "";
|
||||
for (let i = 0; i < parts.length; i++) {
|
||||
const p = parts[i];
|
||||
const filePath = path.join(dir, p.fileName);
|
||||
if (i === 0) firstPath = filePath;
|
||||
const fh = await open(filePath, "w");
|
||||
try {
|
||||
await fh.truncate(p.size); // create the sparse hole
|
||||
for (const r of p.regions) {
|
||||
await fh.write(r.bytes, 0, r.bytes.length, r.offset);
|
||||
}
|
||||
} finally {
|
||||
await fh.close();
|
||||
}
|
||||
}
|
||||
const entries = await lister(firstPath);
|
||||
return entries.length > 0 ? entries : null;
|
||||
} catch (err) {
|
||||
log.warn({ err }, "sparse listing failed");
|
||||
return null;
|
||||
} finally {
|
||||
await rm(dir, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,6 @@
|
||||
import { db } from "./client.js";
|
||||
import type { ArchiveType, FetchStatus } from "@prisma/client";
|
||||
import type { FileEntry } from "../archive/zip-reader.js";
|
||||
|
||||
export async function getActiveAccounts() {
|
||||
return db.telegramAccount.findMany({
|
||||
@@ -375,6 +376,8 @@ export interface ActivityUpdate {
|
||||
zipsFound?: number;
|
||||
zipsDuplicate?: number;
|
||||
zipsIngested?: number;
|
||||
zipsBackfilled?: number;
|
||||
zipsForwarded?: number;
|
||||
}
|
||||
|
||||
export async function updateRunActivity(
|
||||
@@ -398,6 +401,8 @@ export async function updateRunActivity(
|
||||
...(activity.zipsFound !== undefined && { zipsFound: activity.zipsFound }),
|
||||
...(activity.zipsDuplicate !== undefined && { zipsDuplicate: activity.zipsDuplicate }),
|
||||
...(activity.zipsIngested !== undefined && { zipsIngested: activity.zipsIngested }),
|
||||
...(activity.zipsBackfilled !== undefined && { zipsBackfilled: activity.zipsBackfilled }),
|
||||
...(activity.zipsForwarded !== undefined && { zipsForwarded: activity.zipsForwarded }),
|
||||
...(activity.currentTopicId !== undefined && { currentTopicId: activity.currentTopicId }),
|
||||
...(activity.currentAccountChannelMapId !== undefined && {
|
||||
currentAccountChannelMapId: activity.currentAccountChannelMapId,
|
||||
@@ -428,6 +433,8 @@ export async function completeIngestionRun(
|
||||
zipsFound: number;
|
||||
zipsDuplicate: number;
|
||||
zipsIngested: number;
|
||||
zipsBackfilled: number;
|
||||
zipsForwarded: number;
|
||||
}
|
||||
) {
|
||||
return db.ingestionRun.update({
|
||||
@@ -635,6 +642,13 @@ export async function setChannelForum(channelId: string, isForum: boolean) {
|
||||
});
|
||||
}
|
||||
|
||||
export async function setChannelAllowsForwarding(channelId: string, allowsForwarding: boolean) {
|
||||
return db.telegramChannel.update({
|
||||
where: { id: channelId },
|
||||
data: { allowsForwarding },
|
||||
});
|
||||
}
|
||||
|
||||
export async function getTopicProgress(mappingId: string) {
|
||||
return db.topicProgress.findMany({
|
||||
where: { accountChannelMapId: mappingId },
|
||||
@@ -1005,3 +1019,167 @@ export async function createAutoGroup(input: {
|
||||
|
||||
return group.id;
|
||||
}
|
||||
|
||||
// ── Provenance backfill ──
|
||||
|
||||
export interface PlaceholderCandidate {
|
||||
id: string;
|
||||
archiveType: string;
|
||||
fileName: string;
|
||||
fileCount: number;
|
||||
fileSize: bigint;
|
||||
destMessageId: bigint | null;
|
||||
destMessageIds: bigint[];
|
||||
destChannel: { telegramId: bigint } | null;
|
||||
}
|
||||
|
||||
type PlaceholderRow = {
|
||||
id: string; archiveType: string; fileName: string; fileCount: number; fileSize: bigint;
|
||||
destMessageId: bigint | null; destMessageIds: bigint[]; destChannelId: string | null;
|
||||
};
|
||||
|
||||
async function enrichWithDestChannel(rows: PlaceholderRow[]): Promise<PlaceholderCandidate[]> {
|
||||
if (rows.length === 0) return [];
|
||||
const destChannelIds = [...new Set(rows.map((r) => r.destChannelId).filter((id): id is string => !!id))];
|
||||
const channels = destChannelIds.length
|
||||
? await db.telegramChannel.findMany({
|
||||
where: { id: { in: destChannelIds } },
|
||||
select: { id: true, telegramId: true },
|
||||
})
|
||||
: [];
|
||||
const telegramIdById = new Map(channels.map((c) => [c.id, c.telegramId]));
|
||||
return rows.map((row) => ({
|
||||
id: row.id, archiveType: row.archiveType, fileName: row.fileName, fileCount: row.fileCount, fileSize: row.fileSize,
|
||||
destMessageId: row.destMessageId, destMessageIds: row.destMessageIds,
|
||||
destChannel: row.destChannelId && telegramIdById.has(row.destChannelId)
|
||||
? { telegramId: telegramIdById.get(row.destChannelId)! }
|
||||
: null,
|
||||
}));
|
||||
}
|
||||
|
||||
/**
|
||||
* Find every placeholder Package matching name+size (oldest first). Package
|
||||
* has no direct `destChannel` relation (only the scalar `destChannelId`), so
|
||||
* each row's destination TelegramChannel telegramId is resolved with a
|
||||
* follow-up lookup rather than a Prisma include.
|
||||
*/
|
||||
export async function findPlaceholderCandidates(
|
||||
destChannelId: string,
|
||||
fileName: string,
|
||||
fileSize: bigint,
|
||||
): Promise<PlaceholderCandidate[]> {
|
||||
const rows = await db.package.findMany({
|
||||
where: {
|
||||
fileName,
|
||||
fileSize,
|
||||
destMessageId: { not: null },
|
||||
// Placeholder provenance (spec §1): manual-upload (source == destination)
|
||||
// OR rebuild record (sourceMessageId == 0 "unknown" sentinel).
|
||||
OR: [
|
||||
{ sourceChannelId: destChannelId },
|
||||
{ sourceMessageId: 0n },
|
||||
],
|
||||
},
|
||||
select: {
|
||||
id: true, archiveType: true, fileName: true, fileCount: true, fileSize: true,
|
||||
destMessageId: true, destMessageIds: true, destChannelId: true,
|
||||
},
|
||||
orderBy: { indexedAt: "asc" },
|
||||
});
|
||||
return enrichWithDestChannel(rows);
|
||||
}
|
||||
|
||||
/**
|
||||
* Find every uploaded Package (any provenance, any channel) matching
|
||||
* name+size, for the forward-priority path's cross-channel CRC-fingerprint
|
||||
* dedup check. Unlike findPlaceholderCandidates, this is NOT restricted to
|
||||
* placeholder rows — it exists to catch the case where the exact same
|
||||
* archive was independently uploaded (not reposted/forwarded) to two
|
||||
* different source channels.
|
||||
*/
|
||||
export async function findFingerprintDedupCandidates(
|
||||
fileName: string,
|
||||
fileSize: bigint,
|
||||
): Promise<PlaceholderCandidate[]> {
|
||||
const rows = await db.package.findMany({
|
||||
where: { fileName, fileSize, destMessageId: { not: null } },
|
||||
select: {
|
||||
id: true, archiveType: true, fileName: true, fileCount: true, fileSize: true,
|
||||
destMessageId: true, destMessageIds: true, destChannelId: true,
|
||||
},
|
||||
orderBy: { indexedAt: "asc" },
|
||||
});
|
||||
return enrichWithDestChannel(rows);
|
||||
}
|
||||
|
||||
export async function findPlaceholderCandidate(
|
||||
destChannelId: string,
|
||||
fileName: string,
|
||||
fileSize: bigint,
|
||||
): Promise<PlaceholderCandidate | null> {
|
||||
return (await findPlaceholderCandidates(destChannelId, fileName, fileSize))[0] ?? null;
|
||||
}
|
||||
|
||||
export async function getPackageFileCrcs(packageId: string): Promise<(string | null)[]> {
|
||||
const rows = await db.packageFile.findMany({
|
||||
where: { packageId },
|
||||
select: { crc32: true },
|
||||
});
|
||||
return rows.map((r) => r.crc32);
|
||||
}
|
||||
|
||||
export interface BackfillProvenanceInput {
|
||||
packageId: string;
|
||||
destChannelId: string; // to re-check placeholder status in-txn
|
||||
sourceChannelId: string;
|
||||
sourceMessageId: bigint;
|
||||
sourceTopicId: bigint | null;
|
||||
sourceCaption: string | null;
|
||||
remoteUniqueId: string | null;
|
||||
creator: string | null; // always set (re-derived by caller)
|
||||
entries?: FileEntry[]; // set only if candidate had fileCount === 0
|
||||
previewData?: Buffer | null; // set only if provided and candidate lacks one
|
||||
previewMsgId?: bigint | null;
|
||||
}
|
||||
|
||||
export async function backfillProvenance(input: BackfillProvenanceInput): Promise<boolean> {
|
||||
return db.$transaction(async (tx) => {
|
||||
const current = await tx.package.findUnique({
|
||||
where: { id: input.packageId },
|
||||
select: { sourceChannelId: true, sourceMessageId: true, previewData: true, fileCount: true },
|
||||
});
|
||||
// Re-check placeholder status inside the txn (another worker may have won).
|
||||
// Placeholder = manual-upload (source==dest) OR rebuild (sourceMessageId==0).
|
||||
const stillPlaceholder =
|
||||
!!current &&
|
||||
(current.sourceChannelId === input.destChannelId || current.sourceMessageId === 0n);
|
||||
if (!stillPlaceholder) return false;
|
||||
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const data: any = {
|
||||
sourceChannelId: input.sourceChannelId,
|
||||
sourceMessageId: input.sourceMessageId,
|
||||
sourceTopicId: input.sourceTopicId,
|
||||
sourceCaption: input.sourceCaption,
|
||||
remoteUniqueId: input.remoteUniqueId,
|
||||
creator: input.creator,
|
||||
};
|
||||
if (input.entries && current.fileCount === 0) {
|
||||
await tx.packageFile.deleteMany({ where: { packageId: input.packageId } });
|
||||
await tx.packageFile.createMany({
|
||||
data: input.entries.map((e) => ({
|
||||
packageId: input.packageId,
|
||||
path: e.path, fileName: e.fileName, extension: e.extension,
|
||||
compressedSize: e.compressedSize, uncompressedSize: e.uncompressedSize, crc32: e.crc32,
|
||||
})),
|
||||
});
|
||||
data.fileCount = input.entries.length;
|
||||
}
|
||||
if (input.previewData && !current.previewData) {
|
||||
data.previewData = input.previewData;
|
||||
data.previewMsgId = input.previewMsgId ?? null;
|
||||
}
|
||||
await tx.package.update({ where: { id: input.packageId }, data });
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
@@ -0,0 +1,261 @@
|
||||
import { db } from "./db/client.js";
|
||||
import { childLogger } from "./util/logger.js";
|
||||
import { invokeWithTimeout } from "./tdlib/download.js";
|
||||
import { fingerprintsMatch, crcFingerprint } from "./archive/fingerprint.js";
|
||||
import {
|
||||
findPlaceholderCandidates,
|
||||
getPackageFileCrcs,
|
||||
backfillProvenance,
|
||||
type PlaceholderCandidate,
|
||||
} from "./db/queries.js";
|
||||
import type { FileEntry } from "./archive/zip-reader.js";
|
||||
import { readSevenZListingRanged, type RangedPart } from "./archive/ranged/sevenz-ranged.js";
|
||||
import { readRarListingRanged } from "./archive/ranged/rar-ranged.js";
|
||||
import { tdlibRangeReader } from "./archive/ranged/range-reader.js";
|
||||
import { fullDownloadListing } from "./archive/ranged/fallback.js";
|
||||
import { readScannedZipListing, readScannedListingRanged } from "./archive/ranged/dispatch.js";
|
||||
import type { Client } from "tdl";
|
||||
|
||||
const log = childLogger("provenance-backfill");
|
||||
|
||||
export interface BackfillArgs {
|
||||
client: Client;
|
||||
destChannelId: string;
|
||||
scannedSourceChannelId: string;
|
||||
fileName: string;
|
||||
fileSize: bigint;
|
||||
archiveType: string;
|
||||
sourceMessageId: bigint;
|
||||
sourceTopicId: bigint | null;
|
||||
sourceCaption: string | null;
|
||||
remoteUniqueId: string | null;
|
||||
creator: string | null;
|
||||
scannedParts: RangedPart[];
|
||||
previewData?: Buffer | null;
|
||||
previewMsgId?: bigint | null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the destination copy's message(s) into ranged parts (file id +
|
||||
* size + name), in order, so a multipart destination copy is reconstructed
|
||||
* with correct per-part sizes and names (the last message carries the
|
||||
* EOCD-bearing tail part for ZIP; multipart RAR needs correctly-suffixed
|
||||
* `.partN.rar` names for `unrar` sibling discovery). Cheap-only: any TDLib
|
||||
* failure here degrades the caller to name-size confidence rather than
|
||||
* falling back to a full download.
|
||||
*/
|
||||
async function resolveDestParts(
|
||||
client: Client,
|
||||
destChatTelegramId: bigint,
|
||||
destMessageIds: bigint[],
|
||||
destMessageId: bigint | null,
|
||||
fallbackFileName: string,
|
||||
): Promise<RangedPart[] | null> {
|
||||
const messageIds = destMessageIds.length > 0 ? destMessageIds : destMessageId ? [destMessageId] : [];
|
||||
if (messageIds.length === 0) return null;
|
||||
try {
|
||||
const parts: RangedPart[] = [];
|
||||
for (const msgId of messageIds) {
|
||||
const msg = (await invokeWithTimeout(client, {
|
||||
_: "getMessage",
|
||||
chat_id: Number(destChatTelegramId),
|
||||
message_id: Number(msgId),
|
||||
})) as { content?: { document?: { document?: { id: number; size?: number }; file_name?: string } } };
|
||||
const doc = msg?.content?.document?.document;
|
||||
if (!doc?.id) return null;
|
||||
const fileName = msg?.content?.document?.file_name || fallbackFileName;
|
||||
parts.push({ fileId: String(doc.id), fileSize: BigInt(doc.size ?? 0), fileName });
|
||||
}
|
||||
return parts;
|
||||
} catch (err) {
|
||||
log.warn({ err, destMessageIds: messageIds.map(Number) }, "destination archive part resolution failed");
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the CRC fingerprint entries for a placeholder candidate: start from
|
||||
* its stored PackageFile CRCs, and if those are incomplete (e.g. a rebuild
|
||||
* candidate with fileCount === 0), fall back to a fresh ranged read of the
|
||||
* candidate's own copy in the destination channel (Task 9).
|
||||
*/
|
||||
export async function resolveCandidateFingerprintEntries(
|
||||
client: Client,
|
||||
candidate: PlaceholderCandidate,
|
||||
): Promise<FileEntry[]> {
|
||||
const candidateCrcs = await getPackageFileCrcs(candidate.id);
|
||||
let candidateEntries: FileEntry[] = candidateCrcs.map((crc) => ({
|
||||
path: "", fileName: "", extension: null, compressedSize: 0n, uncompressedSize: 0n, crc32: crc,
|
||||
}));
|
||||
const hasDestMessage = candidate.destMessageIds.length > 0 || candidate.destMessageId != null;
|
||||
if (!crcFingerprint(candidateEntries).complete && hasDestMessage && candidate.destChannel) {
|
||||
const destParts = await resolveDestParts(
|
||||
client,
|
||||
candidate.destChannel.telegramId,
|
||||
candidate.destMessageIds,
|
||||
candidate.destMessageId,
|
||||
candidate.fileName,
|
||||
);
|
||||
let destEntries: FileEntry[] | null = null;
|
||||
if (destParts) {
|
||||
const read = tdlibRangeReader(client);
|
||||
destEntries =
|
||||
candidate.archiveType === "ZIP" ? await readScannedZipListing(client, destParts)
|
||||
: candidate.archiveType === "SEVEN_Z" ? await readSevenZListingRanged(destParts, read)
|
||||
: candidate.archiveType === "RAR" ? await readRarListingRanged(destParts, read)
|
||||
: null;
|
||||
}
|
||||
if (destEntries) {
|
||||
candidateEntries = destEntries;
|
||||
}
|
||||
}
|
||||
return candidateEntries;
|
||||
}
|
||||
|
||||
/**
|
||||
* Classify a fingerprint comparison between two entry sets. "incomplete"
|
||||
* means at least one side is missing CRCs (e.g. an empty file → CRC32 of
|
||||
* zero-length data → null) and the comparison CANNOT be used to confirm or
|
||||
* refute a match — callers must fall back to name+size confidence rather
|
||||
* than treating this as a mismatch.
|
||||
*/
|
||||
export function compareFingerprints(a: FileEntry[], b: FileEntry[]): "match" | "mismatch" | "incomplete" {
|
||||
const fa = crcFingerprint(a);
|
||||
const fb = crcFingerprint(b);
|
||||
if (!fa.complete || !fb.complete) return "incomplete";
|
||||
return fingerprintsMatch(a, b) ? "match" : "mismatch";
|
||||
}
|
||||
|
||||
export async function tryProvenanceBackfill(
|
||||
args: BackfillArgs,
|
||||
): Promise<{ backfilled: boolean; confidence?: "fingerprint" | "name-size" }> {
|
||||
const candidates = await findPlaceholderCandidates(args.destChannelId, args.fileName, args.fileSize);
|
||||
if (candidates.length === 0) return { backfilled: false };
|
||||
|
||||
let scannedEntries: FileEntry[] | null = await readScannedListingRanged(
|
||||
args.archiveType, args.client, args.scannedParts,
|
||||
);
|
||||
// Cheap ranged read failed — fall back to a size-capped full download so the
|
||||
// listing still gets indexed. Only worth it when the candidate lacks a listing.
|
||||
if (!scannedEntries && candidates.some((c) => c.fileCount === 0)) {
|
||||
const totalSize = args.scannedParts.reduce((s, p) => s + p.fileSize, 0n);
|
||||
scannedEntries = await fullDownloadListing({
|
||||
client: args.client, parts: args.scannedParts, archiveType: args.archiveType,
|
||||
totalSize, fileName: args.fileName,
|
||||
});
|
||||
}
|
||||
|
||||
let chosen = candidates[0];
|
||||
let confidence: "fingerprint" | "name-size" = "name-size";
|
||||
|
||||
if (candidates.length > 1) {
|
||||
// Multiple placeholder packages share this name+size. Try to
|
||||
// disambiguate by fingerprint (ZIP only); if we can't uniquely resolve
|
||||
// it, notify instead of guessing which one is the real match.
|
||||
if (args.archiveType === "ZIP" && scannedEntries) {
|
||||
const matches: PlaceholderCandidate[] = [];
|
||||
// Candidates NOT ruled out as a definite (both-complete) mismatch —
|
||||
// used as the name+size fallback pool when the fingerprint can't
|
||||
// confirm a match (e.g. incomplete CRCs on either side).
|
||||
const nonMismatches: PlaceholderCandidate[] = [];
|
||||
for (const c of candidates) {
|
||||
const candidateEntries = await resolveCandidateFingerprintEntries(args.client, c);
|
||||
const comparison = compareFingerprints(scannedEntries, candidateEntries);
|
||||
if (comparison === "match") {
|
||||
matches.push(c);
|
||||
nonMismatches.push(c);
|
||||
} else if (comparison === "incomplete") {
|
||||
nonMismatches.push(c);
|
||||
}
|
||||
// comparison === "mismatch": both sides complete and differ — excluded.
|
||||
}
|
||||
if (matches.length === 1) {
|
||||
chosen = matches[0];
|
||||
confidence = "fingerprint";
|
||||
} else if (matches.length === 0 && nonMismatches.length === 1) {
|
||||
// Fingerprint couldn't confirm (incomplete CRCs), but exactly one
|
||||
// candidate wasn't ruled out as a definite mismatch — fall back to
|
||||
// name+size confidence rather than treating this as unresolved.
|
||||
chosen = nonMismatches[0];
|
||||
confidence = "name-size";
|
||||
} else {
|
||||
await db.systemNotification.create({
|
||||
data: {
|
||||
type: "INTEGRITY_AUDIT",
|
||||
severity: "WARNING",
|
||||
title: `Ambiguous provenance match: ${args.fileName}`,
|
||||
message: `${candidates.length} placeholder packages share this name+size and the fingerprint did not uniquely disambiguate. No provenance was backfilled.`,
|
||||
context: { fileName: args.fileName, candidateIds: candidates.map((c) => c.id) },
|
||||
},
|
||||
});
|
||||
return { backfilled: false };
|
||||
}
|
||||
} else {
|
||||
// Can't disambiguate without a fingerprint — notify, don't guess.
|
||||
await db.systemNotification.create({
|
||||
data: {
|
||||
type: "INTEGRITY_AUDIT",
|
||||
severity: "WARNING",
|
||||
title: `Ambiguous provenance match: ${args.fileName}`,
|
||||
message: `${candidates.length} placeholder packages share this name+size (archive type ${args.archiveType} — no cheap fingerprint). No provenance was backfilled.`,
|
||||
context: { fileName: args.fileName, candidateIds: candidates.map((c) => c.id) },
|
||||
},
|
||||
});
|
||||
return { backfilled: false };
|
||||
}
|
||||
} else if (scannedEntries) {
|
||||
const candidateEntries = await resolveCandidateFingerprintEntries(args.client, chosen);
|
||||
const comparison = compareFingerprints(scannedEntries, candidateEntries);
|
||||
if (comparison === "match") {
|
||||
confidence = "fingerprint";
|
||||
} else if (comparison === "mismatch") {
|
||||
// Both sides' CRCs are complete and differ: NOT the same content
|
||||
// despite name+size. Do not backfill.
|
||||
log.info({ candidateId: chosen.id, fileName: args.fileName }, "fingerprint mismatch — not backfilling");
|
||||
return { backfilled: false };
|
||||
}
|
||||
// comparison === "incomplete": can't confirm or refute by fingerprint —
|
||||
// fall through and backfill on name+size confidence instead.
|
||||
}
|
||||
|
||||
const ok = await backfillProvenance({
|
||||
packageId: chosen.id,
|
||||
destChannelId: args.destChannelId,
|
||||
sourceChannelId: args.scannedSourceChannelId,
|
||||
sourceMessageId: args.sourceMessageId,
|
||||
sourceTopicId: args.sourceTopicId,
|
||||
sourceCaption: args.sourceCaption,
|
||||
remoteUniqueId: args.remoteUniqueId,
|
||||
creator: args.creator,
|
||||
entries: chosen.fileCount === 0 && scannedEntries ? scannedEntries : undefined,
|
||||
previewData: args.previewData ?? undefined,
|
||||
previewMsgId: args.previewMsgId ?? undefined,
|
||||
});
|
||||
|
||||
if (!ok) return { backfilled: false };
|
||||
|
||||
if (confidence === "name-size") {
|
||||
// Lower-confidence backfill: no CRC fingerprint guard confirmed this
|
||||
// match. Record it as an auditable event so name+size-only backfills
|
||||
// can be reviewed after the fact.
|
||||
await db.systemNotification.create({
|
||||
data: {
|
||||
type: "INTEGRITY_AUDIT",
|
||||
severity: "INFO",
|
||||
title: `Provenance backfilled by name+size: ${args.fileName}`,
|
||||
message: `Package ${chosen.id} was matched to a scanned source message by file name and size only (no CRC fingerprint confirmation).`,
|
||||
context: {
|
||||
packageId: chosen.id,
|
||||
fileName: args.fileName,
|
||||
sourceChannelId: args.scannedSourceChannelId,
|
||||
},
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
log.info(
|
||||
{ candidateId: chosen.id, fileName: args.fileName, confidence, source: args.scannedSourceChannelId },
|
||||
"provenance backfilled",
|
||||
);
|
||||
return { backfilled: true, confidence };
|
||||
}
|
||||
@@ -0,0 +1,57 @@
|
||||
import { open } from "fs/promises";
|
||||
import type { Client } from "tdl";
|
||||
import { childLogger } from "../util/logger.js";
|
||||
import { withFloodWait } from "../util/retry.js";
|
||||
|
||||
const log = childLogger("range-download");
|
||||
const RANGE_TIMEOUT_MS = 120_000;
|
||||
|
||||
// NOTE (from Task 4 spike): TDLib writes the requested region into
|
||||
// file.local.path at its ABSOLUTE file offset; file.local.downloaded_prefix_size
|
||||
// counts contiguous bytes from download_offset. We request a 1KB-aligned offset
|
||||
// so downloaded_prefix_size covers our whole [offset, offset+limit) window.
|
||||
// This absolute-offset assumption is PENDING LIVE VERIFICATION ON DEPLOY —
|
||||
// the authenticated TDLib session could not be spiked in this environment.
|
||||
export async function downloadFileRange(
|
||||
client: Client,
|
||||
fileId: string,
|
||||
offset: number,
|
||||
limit: number,
|
||||
expectedSize: bigint,
|
||||
): Promise<Buffer> {
|
||||
const numericId = parseInt(fileId, 10);
|
||||
const alignedOffset = Math.max(0, offset - (offset % 1024));
|
||||
const alignedLimit = limit + (offset - alignedOffset);
|
||||
|
||||
const file = await withFloodWait(
|
||||
() =>
|
||||
new Promise<{ local: { path: string; download_offset: number; downloaded_prefix_size: number } }>(
|
||||
(resolve, reject) => {
|
||||
const timer = setTimeout(() => reject(new Error(`Range download timed out for ${fileId}`)), RANGE_TIMEOUT_MS);
|
||||
client
|
||||
.invoke({
|
||||
_: "downloadFile",
|
||||
file_id: numericId,
|
||||
priority: 1,
|
||||
offset: alignedOffset,
|
||||
limit: alignedLimit,
|
||||
synchronous: true,
|
||||
} as never)
|
||||
.then((f) => { clearTimeout(timer); resolve(f as never); })
|
||||
.catch((e) => { clearTimeout(timer); reject(e); });
|
||||
},
|
||||
),
|
||||
`downloadFileRange:${fileId}`,
|
||||
);
|
||||
|
||||
const start = offset;
|
||||
const fh = await open(file.local.path, "r");
|
||||
try {
|
||||
const buf = Buffer.alloc(limit);
|
||||
const { bytesRead } = await fh.read(buf, 0, limit, start);
|
||||
log.debug({ fileId, offset, limit, bytesRead }, "range read");
|
||||
return bytesRead < limit ? buf.subarray(0, bytesRead) : buf;
|
||||
} finally {
|
||||
await fh.close();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
import { describe, it, expect, vi } from "vitest";
|
||||
import { forwardArchiveToChannel } from "./forward.js";
|
||||
|
||||
function fakeClient(response: unknown) {
|
||||
return { invoke: vi.fn(async () => response) } as never;
|
||||
}
|
||||
|
||||
describe("forwardArchiveToChannel", () => {
|
||||
it("sorts message ids ascending and sends them via forwardMessages", async () => {
|
||||
const invoke = vi.fn(async (req: { message_ids: number[] }) => ({
|
||||
messages: req.message_ids.map((id) => ({ id: id + 1000 })),
|
||||
}));
|
||||
const client = { invoke } as never;
|
||||
|
||||
const result = await forwardArchiveToChannel(client, 111n, 222n, [30n, 10n, 20n]);
|
||||
|
||||
expect(invoke).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
_: "forwardMessages",
|
||||
chat_id: 222,
|
||||
from_chat_id: 111,
|
||||
message_ids: [10, 20, 30],
|
||||
send_copy: false,
|
||||
}),
|
||||
);
|
||||
expect(result.messageId).toBe(1010n);
|
||||
expect(result.messageIds).toEqual([1010n, 1020n, 1030n]);
|
||||
});
|
||||
|
||||
it("throws when Telegram returns null for a message (can't be forwarded)", async () => {
|
||||
const client = fakeClient({ messages: [{ id: 1001 }, null] });
|
||||
await expect(forwardArchiveToChannel(client, 111n, 222n, [10n, 20n])).rejects.toThrow(/could not forward/);
|
||||
});
|
||||
|
||||
it("throws when the response has the wrong number of messages", async () => {
|
||||
const client = fakeClient({ messages: [{ id: 1001 }] });
|
||||
await expect(forwardArchiveToChannel(client, 111n, 222n, [10n, 20n])).rejects.toThrow(/expected 2/);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,73 @@
|
||||
import type { Client } from "tdl";
|
||||
import { childLogger } from "../util/logger.js";
|
||||
import { withFloodWait } from "../util/retry.js";
|
||||
|
||||
const log = childLogger("forward");
|
||||
|
||||
export interface ForwardResult {
|
||||
messageId: bigint;
|
||||
messageIds: bigint[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Forward all parts of an archive set from the source chat directly to the
|
||||
* destination chat via TDLib's forwardMessages — no download, no re-upload.
|
||||
* Only usable when the source channel allows forwarding
|
||||
* (TelegramChannel.allowsForwarding); the caller is responsible for that
|
||||
* check. message_ids must be in strictly increasing order per the TDLib API,
|
||||
* so this always sorts them regardless of the order they're passed in.
|
||||
*/
|
||||
export async function forwardArchiveToChannel(
|
||||
client: Client,
|
||||
fromChatId: bigint,
|
||||
toChatId: bigint,
|
||||
sourceMessageIds: bigint[],
|
||||
): Promise<ForwardResult> {
|
||||
const sortedIds = [...sourceMessageIds].sort((a, b) => (a < b ? -1 : a > b ? 1 : 0));
|
||||
const numericIds = sortedIds.map((id) => Number(id));
|
||||
|
||||
log.info(
|
||||
{ fromChatId: Number(fromChatId), toChatId: Number(toChatId), count: numericIds.length },
|
||||
"Forwarding archive to destination channel"
|
||||
);
|
||||
|
||||
const result = (await withFloodWait(
|
||||
() =>
|
||||
client.invoke({
|
||||
_: "forwardMessages",
|
||||
chat_id: Number(toChatId),
|
||||
topic_id: null,
|
||||
from_chat_id: Number(fromChatId),
|
||||
message_ids: numericIds,
|
||||
options: null,
|
||||
send_copy: false,
|
||||
remove_caption: false,
|
||||
} as never),
|
||||
"forwardMessages"
|
||||
)) as { messages: ({ id: number } | null)[] };
|
||||
|
||||
const forwarded = result.messages;
|
||||
if (!forwarded || forwarded.length !== numericIds.length) {
|
||||
throw new Error(
|
||||
`forwardMessages returned ${forwarded?.length ?? 0} messages, expected ${numericIds.length}`
|
||||
);
|
||||
}
|
||||
|
||||
const messageIds: bigint[] = [];
|
||||
for (let i = 0; i < forwarded.length; i++) {
|
||||
const msg = forwarded[i];
|
||||
if (!msg) {
|
||||
throw new Error(
|
||||
`forwardMessages could not forward source message ${sortedIds[i]} (Telegram returned null — message may not be forwardable)`
|
||||
);
|
||||
}
|
||||
messageIds.push(BigInt(msg.id));
|
||||
}
|
||||
|
||||
log.info(
|
||||
{ fromChatId: Number(fromChatId), toChatId: Number(toChatId), messageIds: messageIds.map(Number) },
|
||||
"Forward confirmed by Telegram"
|
||||
);
|
||||
|
||||
return { messageId: messageIds[0], messageIds };
|
||||
}
|
||||
+284
-32
@@ -16,6 +16,7 @@ import {
|
||||
updateLastProcessedMessage,
|
||||
updateRunActivity,
|
||||
setChannelForum,
|
||||
setChannelAllowsForwarding,
|
||||
getTopicProgress,
|
||||
upsertTopicProgress,
|
||||
upsertChannel,
|
||||
@@ -63,9 +64,14 @@ import { hashParts } from "./archive/hash.js";
|
||||
import { readZipCentralDirectory } from "./archive/zip-reader.js";
|
||||
import { readRarContents } from "./archive/rar-reader.js";
|
||||
import { read7zContents } from "./archive/sevenz-reader.js";
|
||||
import { readScannedListingRanged } from "./archive/ranged/dispatch.js";
|
||||
import { deriveForwardContentHash } from "./archive/forward-identity.js";
|
||||
import { checkFingerprintRepost } from "./archive/forward-repost-check.js";
|
||||
import { forwardArchiveToChannel } from "./upload/forward.js";
|
||||
import { tryProvenanceBackfill } from "./provenance-backfill.js";
|
||||
import { byteLevelSplit, concatenateFiles } from "./archive/split.js";
|
||||
import { uploadToChannel, UploadStallError } from "./upload/channel.js";
|
||||
import { processAlbumGroups, processRuleBasedGroups, processTimeWindowGroups, processPatternGroups, processCreatorGroups, processZipPathGroups, processReplyChainGroups, processCaptionGroups, detectGroupingConflicts, type IndexedPackageRef } from "./grouping.js";
|
||||
import { processAlbumGroups, detectGroupingConflicts, type IndexedPackageRef } from "./grouping.js";
|
||||
import { db } from "./db/client.js";
|
||||
import type { TelegramAccount, TelegramChannel } from "@prisma/client";
|
||||
import type { Client } from "tdl";
|
||||
@@ -314,6 +320,8 @@ interface PipelineContext {
|
||||
zipsFound: number;
|
||||
zipsDuplicate: number;
|
||||
zipsIngested: number;
|
||||
zipsBackfilled: number;
|
||||
zipsForwarded: number;
|
||||
};
|
||||
/** Creator from forum topic name (null for non-forum). */
|
||||
topicCreator: string | null;
|
||||
@@ -422,6 +430,8 @@ export async function runWorkerForAccount(
|
||||
zipsFound: 0,
|
||||
zipsDuplicate: 0,
|
||||
zipsIngested: 0,
|
||||
zipsBackfilled: 0,
|
||||
zipsForwarded: 0,
|
||||
};
|
||||
|
||||
try {
|
||||
@@ -490,9 +500,13 @@ export async function runWorkerForAccount(
|
||||
try {
|
||||
// ── Ensure TDLib knows about this chat ──
|
||||
// getChats may not have loaded all channels (pagination, archive folder, etc.)
|
||||
// so we explicitly load each channel before scanning.
|
||||
// so we explicitly load each channel before scanning. The response is
|
||||
// also where we read has_protected_content (below) to decide whether
|
||||
// this channel is eligible for the forward-priority ingestion path.
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
let chatInfo: any;
|
||||
try {
|
||||
await client.invoke({
|
||||
chatInfo = await client.invoke({
|
||||
_: "getChat",
|
||||
chat_id: Number(channel.telegramId),
|
||||
});
|
||||
@@ -514,6 +528,28 @@ export async function runWorkerForAccount(
|
||||
);
|
||||
}
|
||||
|
||||
// ── Check if channel allows forwarding ──
|
||||
// TDLib's chat.has_protected_content is documented on the general
|
||||
// Chat object (core.telegram.org/tdlib/docs/classtd_1_1td__api_1_1chat.html),
|
||||
// but PENDING LIVE VERIFICATION here: confirm on first deploy that a
|
||||
// real chatTypeSupergroup/channel response actually populates this
|
||||
// field (some TDLib doc pages describe it in the context of basic
|
||||
// groups only). If it's ever `undefined` in practice, this block is a
|
||||
// no-op and allowsForwarding stays at its last-known/null value —
|
||||
// which safely keeps the channel on the download path.
|
||||
const hasProtectedContent: boolean | undefined = chatInfo?.has_protected_content;
|
||||
if (typeof hasProtectedContent === "boolean") {
|
||||
const allowsForwarding = !hasProtectedContent;
|
||||
if (allowsForwarding !== channel.allowsForwarding) {
|
||||
await setChannelAllowsForwarding(channel.id, allowsForwarding);
|
||||
accountLog.info(
|
||||
{ channelId: channel.id, title: channel.title, allowsForwarding },
|
||||
"Updated channel forwarding permission"
|
||||
);
|
||||
}
|
||||
channel.allowsForwarding = allowsForwarding;
|
||||
}
|
||||
|
||||
const pipelineCtx: PipelineContext = {
|
||||
client,
|
||||
runId: activeRunId,
|
||||
@@ -1187,7 +1223,7 @@ export async function runWorkerForAccount(
|
||||
*/
|
||||
function inferSkipReason(errMsg: string): "DOWNLOAD_FAILED" | "UPLOAD_FAILED" | "EXTRACT_FAILED" {
|
||||
const lower = errMsg.toLowerCase();
|
||||
if (lower.includes("upload") || lower.includes("too many requests") || lower.includes("retry after") || lower.includes("send")) {
|
||||
if (lower.includes("upload") || lower.includes("forward") || lower.includes("too many requests") || lower.includes("retry after") || lower.includes("send")) {
|
||||
return "UPLOAD_FAILED";
|
||||
}
|
||||
if (lower.includes("extract") || lower.includes("metadata") || lower.includes("central directory") || lower.includes("archive")) {
|
||||
@@ -1479,34 +1515,13 @@ async function processArchiveSets(
|
||||
scanResult.photos
|
||||
);
|
||||
|
||||
// Auto-grouping passes (gated by per-channel flag)
|
||||
const channelRecord = await db.telegramChannel.findUnique({
|
||||
where: { id: channel.id },
|
||||
select: { autoGroupEnabled: true },
|
||||
});
|
||||
|
||||
if (channelRecord?.autoGroupEnabled !== false) {
|
||||
// Learned rule-based grouping (from manual overrides)
|
||||
await processRuleBasedGroups(channel.id, indexedPackageRefs);
|
||||
|
||||
// Time-window grouping for remaining ungrouped packages
|
||||
await processTimeWindowGroups(channel.id, indexedPackageRefs);
|
||||
|
||||
// Pattern-based grouping (date patterns, project slugs)
|
||||
await processPatternGroups(channel.id, indexedPackageRefs);
|
||||
|
||||
// Creator-based grouping (3+ files from same creator)
|
||||
await processCreatorGroups(channel.id, indexedPackageRefs);
|
||||
|
||||
// ZIP path prefix grouping (shared root folder inside archives)
|
||||
await processZipPathGroups(channel.id, indexedPackageRefs);
|
||||
|
||||
// Reply chain grouping (messages replying to same root)
|
||||
await processReplyChainGroups(channel.id, indexedPackageRefs);
|
||||
|
||||
// Caption fuzzy match grouping
|
||||
await processCaptionGroups(channel.id, indexedPackageRefs);
|
||||
}
|
||||
// Heuristic auto-grouping passes (rule/time/pattern/creator/zip-path/
|
||||
// reply-chain/caption) were removed: the STL view is now a flat list
|
||||
// organized by the creator filter, so automatically inventing groups at
|
||||
// ingestion is no longer wanted. Album grouping above is kept because it
|
||||
// reflects real upload structure (files posted together as one Telegram
|
||||
// album), not a heuristic guess. Existing groups and the manual grouping
|
||||
// actions in the UI are unaffected.
|
||||
|
||||
// Check for potential grouping conflicts
|
||||
await detectGroupingConflicts(channel.id, indexedPackageRefs);
|
||||
@@ -1666,6 +1681,56 @@ async function processOneArchiveSet(
|
||||
return null;
|
||||
}
|
||||
|
||||
// ── Cross-channel provenance backfill ──
|
||||
// The same-channel checks above missed. Before downloading, see if this
|
||||
// archive is the true origin of a placeholder-source package (manual upload
|
||||
// / rebuild record whose sourceChannelId == destChannelId). If so, backfill
|
||||
// its real provenance and skip the download entirely.
|
||||
const archType = archiveSet.type === "7Z" ? "SEVEN_Z" : archiveSet.type;
|
||||
if (destChannelId && (archType === "ZIP" || archType === "RAR" || archType === "SEVEN_Z")) {
|
||||
try {
|
||||
const derivedCreator =
|
||||
topicCreator && topicCreator !== "General"
|
||||
? topicCreator
|
||||
: (extractCreatorFromFileName(archiveName) ?? topicCreator ?? null);
|
||||
const preview = previewMatches.get(archiveSet.baseName);
|
||||
const result = await tryProvenanceBackfill({
|
||||
client,
|
||||
destChannelId,
|
||||
scannedSourceChannelId: channel.id,
|
||||
fileName: archiveName,
|
||||
fileSize: totalArchiveSize,
|
||||
archiveType: archType,
|
||||
sourceMessageId: archiveSet.parts[0].id,
|
||||
sourceTopicId,
|
||||
sourceCaption: archiveSet.parts[0].caption ?? null,
|
||||
remoteUniqueId: archiveSet.parts[0].remoteUniqueId ?? null,
|
||||
creator: derivedCreator,
|
||||
scannedParts: archiveSet.parts.map((p) => ({ fileId: p.fileId, fileSize: p.fileSize, fileName: p.fileName })),
|
||||
previewData: null,
|
||||
previewMsgId: preview?.id ?? null,
|
||||
});
|
||||
if (result.backfilled) {
|
||||
counters.zipsBackfilled++;
|
||||
accountLog.info(
|
||||
{ fileName: archiveName, sourceMessageId: Number(archiveSet.parts[0].id), confidence: result.confidence },
|
||||
"Backfilled provenance for placeholder package — skipping download",
|
||||
);
|
||||
await updateRunActivity(runId, {
|
||||
currentActivity: `Backfilled provenance for ${archiveName}`,
|
||||
currentStep: "backfilling",
|
||||
currentFile: archiveName,
|
||||
currentFileNum: setIdx + 1,
|
||||
totalFiles: totalSets,
|
||||
zipsBackfilled: counters.zipsBackfilled,
|
||||
});
|
||||
return null;
|
||||
}
|
||||
} catch (err) {
|
||||
accountLog.warn({ err, fileName: archiveName }, "Provenance backfill attempt failed (non-fatal), continuing to normal ingestion");
|
||||
}
|
||||
}
|
||||
|
||||
// ── Size guard: skip archives that exceed WORKER_MAX_ZIP_SIZE_MB ──
|
||||
const maxSizeBytes = BigInt(config.maxZipSizeMB) * 1024n * 1024n;
|
||||
if (totalArchiveSize > maxSizeBytes) {
|
||||
@@ -1699,6 +1764,31 @@ async function processOneArchiveSet(
|
||||
return null;
|
||||
}
|
||||
|
||||
// ── Forward-priority path ──
|
||||
// If the source channel allows forwarding, try to index + forward without a
|
||||
// local download. Any failure (ranged listing miss, blocked/failed forward)
|
||||
// falls through into the existing download pipeline below so indexing
|
||||
// completeness never regresses.
|
||||
if (channel.allowsForwarding === true) {
|
||||
try {
|
||||
const forwardResult = await tryForwardArchiveSet(
|
||||
ctx, archiveSet, setIdx, totalSets, previewMatches, ingestionRunId
|
||||
);
|
||||
if (forwardResult !== undefined) {
|
||||
return forwardResult;
|
||||
}
|
||||
accountLog.info(
|
||||
{ fileName: archiveName },
|
||||
"Forward path unavailable for this archive — falling back to download+reupload"
|
||||
);
|
||||
} catch (forwardPathErr) {
|
||||
accountLog.warn(
|
||||
{ err: forwardPathErr, fileName: archiveName },
|
||||
"Forward path threw unexpectedly — falling back to download+reupload"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
const tempPaths: string[] = [];
|
||||
let splitPaths: string[] = [];
|
||||
|
||||
@@ -2265,6 +2355,168 @@ async function processOneArchiveSet(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Attempt the forward-priority path for one archive set: ranged listing (no
|
||||
* download) + native Telegram forward to the destination channel.
|
||||
*
|
||||
* Returns `undefined` when the forward path isn't usable for this specific
|
||||
* archive (ranged listing failed, or the forward itself failed) — the caller
|
||||
* falls through to the existing download+reupload pipeline in that case, so
|
||||
* indexing completeness never regresses.
|
||||
*
|
||||
* Returns `null` when the archive is a confirmed duplicate (skip, same
|
||||
* contract as the pre-download dedup checks earlier in the caller).
|
||||
*
|
||||
* Returns the new Package id on success.
|
||||
*/
|
||||
async function tryForwardArchiveSet(
|
||||
ctx: PipelineContext,
|
||||
archiveSet: ArchiveSet,
|
||||
setIdx: number,
|
||||
totalSets: number,
|
||||
previewMatches: Map<string, { id: bigint; fileId: string }>,
|
||||
ingestionRunId: string,
|
||||
): Promise<string | null | undefined> {
|
||||
const {
|
||||
client, channelTitle, channel,
|
||||
destChannelTelegramId, destChannelId,
|
||||
counters, topicCreator, sourceTopicId, accountLog,
|
||||
} = ctx;
|
||||
void setIdx;
|
||||
void totalSets;
|
||||
|
||||
const archiveName = archiveSet.parts[0].fileName;
|
||||
const archType = archiveSet.type === "7Z" ? ("SEVEN_Z" as const) : archiveSet.type;
|
||||
if (archType !== "ZIP" && archType !== "RAR" && archType !== "SEVEN_Z") {
|
||||
// The ranged listing readers only cover archive formats. Standalone
|
||||
// DOCUMENT attachments always go through the existing download path,
|
||||
// which for DOCUMENT is already cheap (no extraction, single entry).
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const scannedParts = archiveSet.parts.map((p) => ({
|
||||
fileId: p.fileId,
|
||||
fileSize: p.fileSize,
|
||||
fileName: p.fileName,
|
||||
}));
|
||||
|
||||
const entries = await readScannedListingRanged(archType, client, scannedParts);
|
||||
if (!entries) return undefined;
|
||||
|
||||
const totalArchiveSize = archiveSet.parts.reduce((sum, p) => sum + p.fileSize, 0n);
|
||||
const firstRemoteUniqueId = archiveSet.parts[0].remoteUniqueId ?? null;
|
||||
const contentHash = deriveForwardContentHash(
|
||||
entries,
|
||||
firstRemoteUniqueId,
|
||||
channel.id,
|
||||
archiveSet.parts[0].id,
|
||||
);
|
||||
|
||||
if (await packageExistsByHash(contentHash)) {
|
||||
counters.zipsDuplicate++;
|
||||
accountLog.debug({ fileName: archiveName, contentHash }, "Forward-path duplicate (hash), skipping");
|
||||
return null;
|
||||
}
|
||||
|
||||
const repost = await checkFingerprintRepost(client, entries, archiveName, totalArchiveSize);
|
||||
if (repost.isDuplicate) {
|
||||
counters.zipsDuplicate++;
|
||||
accountLog.info(
|
||||
{ fileName: archiveName, matchedPackageId: repost.matchedPackageId },
|
||||
"Forward-path duplicate (CRC fingerprint match against another channel's copy), skipping"
|
||||
);
|
||||
return null;
|
||||
}
|
||||
|
||||
const hashLockAcquired = await tryAcquireHashLock(contentHash);
|
||||
if (!hashLockAcquired) {
|
||||
counters.zipsDuplicate++;
|
||||
accountLog.info(
|
||||
{ fileName: archiveName, contentHash },
|
||||
"Hash lock held by another worker — skipping concurrent duplicate"
|
||||
);
|
||||
return null;
|
||||
}
|
||||
|
||||
try {
|
||||
if (await packageExistsByHash(contentHash)) {
|
||||
counters.zipsDuplicate++;
|
||||
return null;
|
||||
}
|
||||
|
||||
let destResult: { messageId: bigint; messageIds: bigint[] };
|
||||
try {
|
||||
destResult = await forwardArchiveToChannel(
|
||||
client,
|
||||
channel.telegramId,
|
||||
destChannelTelegramId,
|
||||
archiveSet.parts.map((p) => p.id),
|
||||
);
|
||||
} catch (forwardErr) {
|
||||
accountLog.warn(
|
||||
{ err: forwardErr, fileName: archiveName },
|
||||
"Forward failed — falling back to download+reupload for this archive"
|
||||
);
|
||||
return undefined;
|
||||
}
|
||||
|
||||
await deleteOrphanedPackageByHash(contentHash);
|
||||
|
||||
const creator =
|
||||
topicCreator ??
|
||||
extractCreatorFromFileName(archiveName) ??
|
||||
extractCreatorFromChannelTitle(channelTitle) ??
|
||||
null;
|
||||
|
||||
const tags: string[] = [];
|
||||
if (channel.category) tags.push(channel.category);
|
||||
for (const tag of extractSlicerTags(entries)) {
|
||||
if (!tags.includes(tag)) tags.push(tag);
|
||||
}
|
||||
|
||||
const stub = await createPackageStub({
|
||||
contentHash,
|
||||
fileName: archiveName,
|
||||
fileSize: totalArchiveSize,
|
||||
archiveType: archType,
|
||||
sourceChannelId: channel.id,
|
||||
sourceMessageId: archiveSet.parts[0].id,
|
||||
sourceTopicId,
|
||||
remoteUniqueId: firstRemoteUniqueId,
|
||||
destChannelId,
|
||||
destMessageId: destResult.messageId,
|
||||
destMessageIds: destResult.messageIds,
|
||||
isMultipart: archiveSet.parts.length > 1,
|
||||
partCount: archiveSet.parts.length,
|
||||
ingestionRunId,
|
||||
creator,
|
||||
tags,
|
||||
});
|
||||
|
||||
counters.zipsForwarded++;
|
||||
await deleteSkippedPackage(channel.id, archiveSet.parts[0].id);
|
||||
|
||||
let previewData: Buffer | null = null;
|
||||
let previewMsgId: bigint | null = null;
|
||||
const matchedPhoto = previewMatches.get(archiveSet.baseName);
|
||||
if (matchedPhoto) {
|
||||
previewData = await downloadPhotoThumbnail(client, matchedPhoto.fileId);
|
||||
if (previewData) previewMsgId = matchedPhoto.id;
|
||||
}
|
||||
|
||||
await updatePackageWithMetadata(stub.id, { files: entries, previewData, previewMsgId });
|
||||
|
||||
accountLog.info(
|
||||
{ fileName: archiveName, contentHash, fileCount: entries.length, creator },
|
||||
"Archive forwarded (no download)"
|
||||
);
|
||||
|
||||
return stub.id;
|
||||
} finally {
|
||||
await releaseHashLock(contentHash);
|
||||
}
|
||||
}
|
||||
|
||||
async function deleteFiles(paths: string[]): Promise<void> {
|
||||
for (const p of paths) {
|
||||
try {
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
import { defineConfig } from "vitest/config";
|
||||
|
||||
export default defineConfig({
|
||||
test: {
|
||||
environment: "node",
|
||||
include: ["src/**/*.test.ts"],
|
||||
},
|
||||
});
|
||||
Reference in New Issue
Block a user