mirror of
https://github.com/xCyanGrizzly/DragonsStash.git
synced 2026-09-21 13:31:42 +00:00
Compare commits
48
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
dadf03212c | ||
|
|
267c72bbe8 | ||
|
|
497c4876a6 | ||
|
|
f5d913eb18 | ||
|
|
1e11dd3fd8 | ||
|
|
086f58f9dd | ||
|
|
3595f6f097 | ||
|
|
abdfa437d9 | ||
|
|
49f14bcb0d | ||
|
|
d4a1cfec99 | ||
|
|
ecedc0fec4 | ||
|
|
1b1f5b7972 | ||
|
|
e822ea3e76 | ||
|
|
9d16156161 | ||
|
|
2e7e6cca9b | ||
|
|
ceae4f384b | ||
|
|
7595543386 | ||
|
|
09ee9da9cc | ||
|
|
7ddf13053f | ||
|
|
4a7b9e2a09 | ||
|
|
9a06130c5b | ||
|
|
a7aa4ce285 | ||
|
|
38072d250f | ||
|
|
0bce1168a9 | ||
|
|
018b0f5d74 | ||
|
|
c2590fb66f | ||
|
|
8b443620c8 | ||
|
|
80aa2b0ee0 | ||
|
|
f0e0e79d34 | ||
|
|
21bd46010f | ||
|
|
2252ac01f5 | ||
|
|
63df348028 | ||
|
|
8bbf51f056 | ||
|
|
5873d148c1 | ||
|
|
90be365c7f | ||
|
|
7e21a41615 | ||
|
|
f3d62c68fb | ||
|
|
412e3066bc | ||
|
|
6178ff3b08 | ||
|
|
57842c6d95 | ||
|
|
b3c49c3794 | ||
|
|
20e60bc6af | ||
|
|
50e89719bb | ||
|
|
1cae855c26 | ||
|
|
b0baf72f0a | ||
|
|
0cf5fcd3a7 | ||
|
|
3b7202a662 | ||
|
|
23f8e91c50 |
+16
-1
@@ -54,9 +54,24 @@ steps:
|
|||||||
password:
|
password:
|
||||||
from_secret: gitea_password
|
from_secret: gitea_password
|
||||||
|
|
||||||
|
- name: build-backup
|
||||||
|
image: plugins/docker
|
||||||
|
depends_on: [clone]
|
||||||
|
settings:
|
||||||
|
repo: git.samagsteribbe.nl/admin/dragonsstash-backup
|
||||||
|
registry: git.samagsteribbe.nl
|
||||||
|
dockerfile: backup/Dockerfile
|
||||||
|
tags:
|
||||||
|
- latest
|
||||||
|
- "${DRONE_COMMIT_SHA:0:8}"
|
||||||
|
username:
|
||||||
|
from_secret: gitea_username
|
||||||
|
password:
|
||||||
|
from_secret: gitea_password
|
||||||
|
|
||||||
- name: deploy
|
- name: deploy
|
||||||
image: alpine
|
image: alpine
|
||||||
depends_on: [build-app, build-worker, build-bot]
|
depends_on: [build-app, build-worker, build-bot, build-backup]
|
||||||
environment:
|
environment:
|
||||||
SSH_KEY:
|
SSH_KEY:
|
||||||
from_secret: ssh_key
|
from_secret: ssh_key
|
||||||
|
|||||||
@@ -36,3 +36,12 @@ TDLIB_STATE_DIR="/data/tdlib"
|
|||||||
WORKER_MAX_ZIP_SIZE_MB=4096
|
WORKER_MAX_ZIP_SIZE_MB=4096
|
||||||
MULTIPART_TIMEOUT_HOURS=0
|
MULTIPART_TIMEOUT_HOURS=0
|
||||||
LOG_LEVEL="info"
|
LOG_LEVEL="info"
|
||||||
|
|
||||||
|
# Backup (NAS via SMB/CIFS + restic)
|
||||||
|
NAS_HOST="" # Synology NAS IP or hostname reachable from this host
|
||||||
|
NAS_SHARE="" # SMB share name, e.g. dragonsstash_backups
|
||||||
|
NAS_USERNAME="" # SMB user with read/write on the share
|
||||||
|
NAS_PASSWORD="" # SMB user password (avoid commas — they delimit cifs mount opts)
|
||||||
|
RESTIC_PASSWORD="" # generate with: openssl rand -base64 32
|
||||||
|
KUMA_PUSH_URL="" # optional: Uptime Kuma Push monitor URL; leave empty to disable alerting
|
||||||
|
TZ="Etc/UTC"
|
||||||
|
|||||||
@@ -0,0 +1,13 @@
|
|||||||
|
FROM alpine:3.20
|
||||||
|
|
||||||
|
# Note: use busybox's built-in crond (Alpine base), NOT the dcron package —
|
||||||
|
# dcron's crond fails with "setpgid: Operation not permitted" in this runtime.
|
||||||
|
RUN apk add --no-cache restic postgresql16-client curl tzdata tar bash
|
||||||
|
|
||||||
|
COPY backup/backup.sh /backup.sh
|
||||||
|
COPY backup/entrypoint.sh /entrypoint.sh
|
||||||
|
COPY backup/crontab /etc/crontabs/root
|
||||||
|
|
||||||
|
RUN chmod +x /backup.sh /entrypoint.sh
|
||||||
|
|
||||||
|
ENTRYPOINT ["/entrypoint.sh"]
|
||||||
@@ -0,0 +1,32 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
report_failure() {
|
||||||
|
[ -n "${KUMA_PUSH_URL:-}" ] || return 0
|
||||||
|
curl -fsS "$KUMA_PUSH_URL" --get \
|
||||||
|
--data-urlencode "status=down" \
|
||||||
|
--data-urlencode "msg=$BASH_COMMAND failed" || true
|
||||||
|
}
|
||||||
|
trap report_failure ERR
|
||||||
|
|
||||||
|
DUMP_FILE=/tmp/dragonsstash.dump
|
||||||
|
TAR_FILE=/tmp/tdlib.tar.gz
|
||||||
|
|
||||||
|
trap 'rm -f "$DUMP_FILE" "$TAR_FILE"' EXIT
|
||||||
|
|
||||||
|
pg_dump -h dragonsstash-db -U "$POSTGRES_USER" -d "$POSTGRES_DB" -Fc -f "$DUMP_FILE"
|
||||||
|
|
||||||
|
# TDLib volumes are tarred live (best-effort, per design). A file changing
|
||||||
|
# mid-read makes GNU tar exit 1 (warning) — that is expected here and must not
|
||||||
|
# abort the backup. Only a genuine error (exit >= 2) is fatal.
|
||||||
|
tar --warning=no-file-changed -czf "$TAR_FILE" -C /data tdlib-worker tdlib-bot \
|
||||||
|
|| { rc=$?; [ "$rc" -le 1 ] || exit "$rc"; }
|
||||||
|
|
||||||
|
restic backup "$DUMP_FILE" "$TAR_FILE"
|
||||||
|
restic forget --keep-daily 14 --prune
|
||||||
|
|
||||||
|
if [ -n "${KUMA_PUSH_URL:-}" ]; then
|
||||||
|
curl -fsS "$KUMA_PUSH_URL" --get \
|
||||||
|
--data-urlencode "status=up" \
|
||||||
|
--data-urlencode "msg=OK"
|
||||||
|
fi
|
||||||
@@ -0,0 +1,2 @@
|
|||||||
|
0 3 * * * /backup.sh >> /proc/1/fd/1 2>&1
|
||||||
|
0 4 * * 0 restic check >> /proc/1/fd/1 2>&1
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
set -uo pipefail
|
||||||
|
|
||||||
|
# Ensure the repo exists, but never crash-loop on it: a transient error reading
|
||||||
|
# the repo (CIFS hiccup, stale lock) must not kill PID 1. `restic init` failing
|
||||||
|
# because the repo already exists is expected and harmless here.
|
||||||
|
if ! restic cat config >/dev/null 2>&1; then
|
||||||
|
restic init || echo "restic init skipped (repo already exists or temporarily unreachable)"
|
||||||
|
fi
|
||||||
|
|
||||||
|
exec crond -f -l 2
|
||||||
+38
-2
@@ -97,6 +97,35 @@ services:
|
|||||||
networks:
|
networks:
|
||||||
- backend
|
- backend
|
||||||
|
|
||||||
|
backup:
|
||||||
|
build:
|
||||||
|
context: .
|
||||||
|
dockerfile: backup/Dockerfile
|
||||||
|
pull_policy: never
|
||||||
|
environment:
|
||||||
|
- POSTGRES_USER=${POSTGRES_USER:-dragons}
|
||||||
|
- POSTGRES_PASSWORD=${POSTGRES_PASSWORD:-stash}
|
||||||
|
- PGPASSWORD=${POSTGRES_PASSWORD:-stash}
|
||||||
|
- POSTGRES_DB=${POSTGRES_DB:-dragonsstash}
|
||||||
|
- RESTIC_REPOSITORY=/backups/restic-repo
|
||||||
|
- RESTIC_PASSWORD=${RESTIC_PASSWORD:?Set RESTIC_PASSWORD in .env}
|
||||||
|
- KUMA_PUSH_URL=${KUMA_PUSH_URL:-}
|
||||||
|
- TZ=${TZ:-Etc/UTC}
|
||||||
|
volumes:
|
||||||
|
- tdlib_state:/data/tdlib-worker:ro
|
||||||
|
- tdlib_bot_state:/data/tdlib-bot:ro
|
||||||
|
- nas_backups:/backups
|
||||||
|
depends_on:
|
||||||
|
db:
|
||||||
|
condition: service_healthy
|
||||||
|
restart: unless-stopped
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
limits:
|
||||||
|
memory: 1G
|
||||||
|
networks:
|
||||||
|
- backend
|
||||||
|
|
||||||
db:
|
db:
|
||||||
image: postgres:16-alpine
|
image: postgres:16-alpine
|
||||||
environment:
|
environment:
|
||||||
@@ -116,8 +145,10 @@ services:
|
|||||||
limits:
|
limits:
|
||||||
memory: 1G
|
memory: 1G
|
||||||
networks:
|
networks:
|
||||||
- frontend
|
frontend: {}
|
||||||
- backend
|
backend:
|
||||||
|
aliases:
|
||||||
|
- dragonsstash-db
|
||||||
|
|
||||||
volumes:
|
volumes:
|
||||||
postgres_data:
|
postgres_data:
|
||||||
@@ -125,6 +156,11 @@ volumes:
|
|||||||
tdlib_bot_state:
|
tdlib_bot_state:
|
||||||
tmp_zips:
|
tmp_zips:
|
||||||
manual_uploads:
|
manual_uploads:
|
||||||
|
nas_backups:
|
||||||
|
driver_opts:
|
||||||
|
type: cifs
|
||||||
|
o: "username=${NAS_USERNAME},password=${NAS_PASSWORD},vers=3.0,uid=0,gid=0,file_mode=0660,dir_mode=0770"
|
||||||
|
device: "//${NAS_HOST}/${NAS_SHARE}"
|
||||||
|
|
||||||
networks:
|
networks:
|
||||||
frontend:
|
frontend:
|
||||||
|
|||||||
@@ -0,0 +1,454 @@
|
|||||||
|
# Send All From Creator Implementation Plan
|
||||||
|
|
||||||
|
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
||||||
|
|
||||||
|
**Goal:** Add a toolbar button on the STL packages page that queues every sendable package from the currently-filtered creator for delivery to the user's Telegram.
|
||||||
|
|
||||||
|
**Architecture:** A new server action `sendAllFromCreatorAction(creatorName)` mirrors the existing `sendAllInGroupAction`: it fetches sendable packages for the creator, dedups against live `BotSendRequest`s, creates a `BotSendRequest` per package, and fires `pg_notify('bot_send', id)`. The bot's existing `send-listener` consumes these unchanged. The client table renders a "Send all from [Creator]" button in the packages-tab toolbar only when `?creator=<name>` is active, wired to the action with a confirm dialog and a count toast.
|
||||||
|
|
||||||
|
**Tech Stack:** Next.js 16 App Router, TypeScript, Prisma v7, server actions, `pg_notify`, sonner toasts, lucide-react icons.
|
||||||
|
|
||||||
|
**Testing note:** This repo has no test framework (`CLAUDE.md`: "Testing is manual"). Verification for each task is `npm run lint` + `npm run build`, plus manual UI checks at the end. Do not scaffold a test harness.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Task 1: Add `sendAllFromCreatorAction` server action
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `src/app/(app)/stls/actions.ts` (append new action after `sendAllInGroupAction`, which ends at line 591)
|
||||||
|
|
||||||
|
Reference implementation to mirror: `sendAllInGroupAction` at `src/app/(app)/stls/actions.ts:515-591`.
|
||||||
|
Existing imports already present in the file (do not re-add): `auth` from `@/lib/auth`, `prisma` from `@/lib/prisma`, `ActionResult` from `@/types/api.types`, `revalidatePath` from `next/cache`.
|
||||||
|
`ActionResult` is generic: `ActionResult<T = void>` (`src/types/api.types.ts`).
|
||||||
|
|
||||||
|
- [ ] **Step 1: Append the new action**
|
||||||
|
|
||||||
|
Add this to the end of `src/app/(app)/stls/actions.ts`:
|
||||||
|
|
||||||
|
```ts
|
||||||
|
export async function sendAllFromCreatorAction(
|
||||||
|
creatorName: string
|
||||||
|
): Promise<ActionResult<{ queued: number; skipped: number }>> {
|
||||||
|
const session = await auth();
|
||||||
|
if (!session?.user?.id) return { success: false, error: "Unauthorized" };
|
||||||
|
|
||||||
|
const creator = creatorName.trim();
|
||||||
|
if (!creator) {
|
||||||
|
return { success: false, error: "No creator specified" };
|
||||||
|
}
|
||||||
|
|
||||||
|
try {
|
||||||
|
const telegramLink = await prisma.telegramLink.findUnique({
|
||||||
|
where: { userId: session.user.id },
|
||||||
|
});
|
||||||
|
|
||||||
|
if (!telegramLink) {
|
||||||
|
return { success: false, error: "No linked Telegram account. Link one in Settings." };
|
||||||
|
}
|
||||||
|
|
||||||
|
const sendablePackages = await prisma.package.findMany({
|
||||||
|
where: {
|
||||||
|
creator,
|
||||||
|
destChannelId: { not: null },
|
||||||
|
destMessageId: { not: null },
|
||||||
|
},
|
||||||
|
select: { id: true },
|
||||||
|
});
|
||||||
|
|
||||||
|
if (sendablePackages.length === 0) {
|
||||||
|
return { success: false, error: "No uploaded packages found for this creator" };
|
||||||
|
}
|
||||||
|
|
||||||
|
let queued = 0;
|
||||||
|
let skipped = 0;
|
||||||
|
for (const pkg of sendablePackages) {
|
||||||
|
// Only create if no existing PENDING/SENDING request for this package+link combo
|
||||||
|
const existing = await prisma.botSendRequest.findFirst({
|
||||||
|
where: {
|
||||||
|
packageId: pkg.id,
|
||||||
|
telegramLinkId: telegramLink.id,
|
||||||
|
status: { in: ["PENDING", "SENDING"] },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
if (existing) {
|
||||||
|
skipped++;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
const sendRequest = await prisma.botSendRequest.create({
|
||||||
|
data: {
|
||||||
|
packageId: pkg.id,
|
||||||
|
telegramLinkId: telegramLink.id,
|
||||||
|
requestedByUserId: session.user.id,
|
||||||
|
status: "PENDING",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
// Notify the bot via pg_notify
|
||||||
|
try {
|
||||||
|
await prisma.$queryRawUnsafe(
|
||||||
|
`SELECT pg_notify('bot_send', $1)`,
|
||||||
|
sendRequest.id
|
||||||
|
);
|
||||||
|
} catch {
|
||||||
|
// Best-effort — the bot also polls periodically
|
||||||
|
}
|
||||||
|
|
||||||
|
queued++;
|
||||||
|
}
|
||||||
|
|
||||||
|
revalidatePath("/stls");
|
||||||
|
return { success: true, data: { queued, skipped } };
|
||||||
|
} catch {
|
||||||
|
return { success: false, error: "Failed to send creator packages" };
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 2: Verify lint + typecheck pass**
|
||||||
|
|
||||||
|
Run: `npm run lint`
|
||||||
|
Expected: no new errors in `src/app/(app)/stls/actions.ts`.
|
||||||
|
|
||||||
|
- [ ] **Step 3: Commit**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
git add src/app/\(app\)/stls/actions.ts
|
||||||
|
git commit -m "feat(stls): add sendAllFromCreatorAction server action"
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Task 2: Wire up the toolbar button in the STL table
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `src/app/(app)/stls/_components/stl-table.tsx`
|
||||||
|
- lucide import (line 6)
|
||||||
|
- actions import block (lines 44-54)
|
||||||
|
- handler (after `handleSendAllInGroup`, which ends at line 278)
|
||||||
|
- `activeCreator` derivation (near `activeTag` at line 445)
|
||||||
|
- toolbar JSX (packages-tab toolbar row starting at line 478)
|
||||||
|
|
||||||
|
- [ ] **Step 1: Add the `Send` icon to the lucide import**
|
||||||
|
|
||||||
|
Change line 6 from:
|
||||||
|
|
||||||
|
```ts
|
||||||
|
import { Search, Layers, Upload } from "lucide-react";
|
||||||
|
```
|
||||||
|
|
||||||
|
to:
|
||||||
|
|
||||||
|
```ts
|
||||||
|
import { Search, Layers, Upload, Send } from "lucide-react";
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 2: Import the new action**
|
||||||
|
|
||||||
|
In the actions import block (lines 44-54), add `sendAllFromCreatorAction` next to `sendAllInGroupAction`:
|
||||||
|
|
||||||
|
```ts
|
||||||
|
import {
|
||||||
|
updatePackageCreator,
|
||||||
|
updatePackageTags,
|
||||||
|
renameGroupAction,
|
||||||
|
dissolveGroupAction,
|
||||||
|
createGroupAction,
|
||||||
|
removeFromGroupAction,
|
||||||
|
sendAllInGroupAction,
|
||||||
|
sendAllFromCreatorAction,
|
||||||
|
updateGroupPreviewAction,
|
||||||
|
mergeGroupsAction,
|
||||||
|
} from "../actions";
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 3: Derive the active creator from the URL**
|
||||||
|
|
||||||
|
Immediately after the `activeTag` line (`src/app/(app)/stls/_components/stl-table.tsx:445`):
|
||||||
|
|
||||||
|
```ts
|
||||||
|
const activeTag = searchParams.get("tag") ?? "";
|
||||||
|
```
|
||||||
|
|
||||||
|
add:
|
||||||
|
|
||||||
|
```ts
|
||||||
|
const activeCreator = searchParams.get("creator") ?? "";
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 4: Add the click handler**
|
||||||
|
|
||||||
|
After `handleSendAllInGroup` (which ends at line 278), add:
|
||||||
|
|
||||||
|
```ts
|
||||||
|
const handleSendAllFromCreator = useCallback(() => {
|
||||||
|
if (!confirm(`Send all packages from "${activeCreator}" to your Telegram?`)) return;
|
||||||
|
startTransition(async () => {
|
||||||
|
const result = await sendAllFromCreatorAction(activeCreator);
|
||||||
|
if (result.success) {
|
||||||
|
const { queued, skipped } = result.data;
|
||||||
|
toast.success(
|
||||||
|
`Queued ${queued} package${queued === 1 ? "" : "s"} from ${activeCreator}` +
|
||||||
|
(skipped ? ` (${skipped} already queued)` : "")
|
||||||
|
);
|
||||||
|
router.refresh();
|
||||||
|
} else {
|
||||||
|
toast.error(result.error);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}, [activeCreator, router]);
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 5: Render the toolbar button**
|
||||||
|
|
||||||
|
In the packages-tab toolbar (`flex flex-wrap items-center gap-2` row at line 478), add the button right after the "Upload Files" button (which closes at line 507, before the `selectedPackages.size >= 2` block at line 508):
|
||||||
|
|
||||||
|
```tsx
|
||||||
|
{activeCreator && (
|
||||||
|
<Button
|
||||||
|
variant="outline"
|
||||||
|
size="sm"
|
||||||
|
className="h-9 gap-1.5"
|
||||||
|
onClick={handleSendAllFromCreator}
|
||||||
|
>
|
||||||
|
<Send className="h-3.5 w-3.5" />
|
||||||
|
Send all from {activeCreator}
|
||||||
|
</Button>
|
||||||
|
)}
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 6: Verify lint + build pass**
|
||||||
|
|
||||||
|
Run: `npm run lint && npm run build`
|
||||||
|
Expected: no new errors; build completes.
|
||||||
|
|
||||||
|
- [ ] **Step 7: Manual verification**
|
||||||
|
|
||||||
|
1. `npm run dev`, open `/stls`.
|
||||||
|
2. Click a creator name in the Creator column to apply `?creator=<name>` (or navigate to `/stls?creator=<known creator>`).
|
||||||
|
3. Confirm the "Send all from [Creator]" button appears in the toolbar.
|
||||||
|
4. Remove the creator filter → confirm the button disappears.
|
||||||
|
5. Click the button → confirm the dialog appears; on confirm, a toast reports the queued count.
|
||||||
|
6. (If a linked Telegram account + uploaded packages exist) confirm packages arrive via the bot.
|
||||||
|
|
||||||
|
- [ ] **Step 8: Commit**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
git add src/app/\(app\)/stls/_components/stl-table.tsx
|
||||||
|
git commit -m "feat(stls): add 'send all from creator' toolbar button"
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Task 3: Searchable creator filter combobox (makes the filter reachable)
|
||||||
|
|
||||||
|
**Why:** Discovered during execution — the `?creator=` filter that reveals the Task 2
|
||||||
|
button had no UI trigger. The creator table cell click only opens the edit prompt
|
||||||
|
(`package-columns.tsx:339` → `onSetCreator`). This task adds a searchable
|
||||||
|
"All Creators" combobox to the packages-tab toolbar (like the existing Tags select,
|
||||||
|
but type-to-filter since there can be many creators). The creator cell stays
|
||||||
|
edit-only.
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `src/lib/telegram/queries.ts` — add `getAllPackageCreators()`.
|
||||||
|
- Modify: `src/app/(app)/stls/page.tsx` — fetch + pass `availableCreators`.
|
||||||
|
- Create: `src/app/(app)/stls/_components/creator-filter.tsx` — the combobox.
|
||||||
|
- Modify: `src/app/(app)/stls/_components/stl-table.tsx` — new prop, handler, render.
|
||||||
|
|
||||||
|
- [ ] **Step 1: Add the distinct-creators query**
|
||||||
|
|
||||||
|
Append to `src/lib/telegram/queries.ts` (mirrors `getAllPackageTags` at line 530):
|
||||||
|
|
||||||
|
```ts
|
||||||
|
export async function getAllPackageCreators(): Promise<string[]> {
|
||||||
|
const result = await prisma.$queryRaw<{ creator: string }[]>`
|
||||||
|
SELECT DISTINCT creator FROM packages
|
||||||
|
WHERE creator IS NOT NULL AND creator <> ''
|
||||||
|
ORDER BY creator
|
||||||
|
`;
|
||||||
|
return result.map((r) => r.creator);
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 2: Create the combobox component**
|
||||||
|
|
||||||
|
Create `src/app/(app)/stls/_components/creator-filter.tsx` (mirrors the
|
||||||
|
Popover+Command pattern in `src/components/shared/data-table-faceted-filter.tsx`):
|
||||||
|
|
||||||
|
```tsx
|
||||||
|
"use client";
|
||||||
|
|
||||||
|
import { useState } from "react";
|
||||||
|
import { Check, ChevronsUpDown } from "lucide-react";
|
||||||
|
import { cn } from "@/lib/utils";
|
||||||
|
import { Button } from "@/components/ui/button";
|
||||||
|
import {
|
||||||
|
Command,
|
||||||
|
CommandEmpty,
|
||||||
|
CommandGroup,
|
||||||
|
CommandInput,
|
||||||
|
CommandItem,
|
||||||
|
CommandList,
|
||||||
|
} from "@/components/ui/command";
|
||||||
|
import { Popover, PopoverContent, PopoverTrigger } from "@/components/ui/popover";
|
||||||
|
|
||||||
|
interface CreatorFilterProps {
|
||||||
|
creators: string[];
|
||||||
|
value: string; // active creator, "" when none
|
||||||
|
onChange: (creator: string) => void; // "" clears the filter
|
||||||
|
}
|
||||||
|
|
||||||
|
export function CreatorFilter({ creators, value, onChange }: CreatorFilterProps) {
|
||||||
|
const [open, setOpen] = useState(false);
|
||||||
|
|
||||||
|
return (
|
||||||
|
<Popover open={open} onOpenChange={setOpen}>
|
||||||
|
<PopoverTrigger asChild>
|
||||||
|
<Button
|
||||||
|
variant="outline"
|
||||||
|
size="sm"
|
||||||
|
role="combobox"
|
||||||
|
aria-expanded={open}
|
||||||
|
className="h-9 w-[200px] justify-between"
|
||||||
|
>
|
||||||
|
<span className="truncate">{value || "All Creators"}</span>
|
||||||
|
<ChevronsUpDown className="ml-2 h-4 w-4 shrink-0 opacity-50" />
|
||||||
|
</Button>
|
||||||
|
</PopoverTrigger>
|
||||||
|
<PopoverContent className="w-[240px] p-0" align="start">
|
||||||
|
<Command>
|
||||||
|
<CommandInput placeholder="Search creators..." className="h-9" />
|
||||||
|
<CommandList>
|
||||||
|
<CommandEmpty>No creators found.</CommandEmpty>
|
||||||
|
<CommandGroup>
|
||||||
|
<CommandItem
|
||||||
|
value="__all__"
|
||||||
|
onSelect={() => {
|
||||||
|
onChange("");
|
||||||
|
setOpen(false);
|
||||||
|
}}
|
||||||
|
>
|
||||||
|
<Check
|
||||||
|
className={cn("mr-2 h-4 w-4", value === "" ? "opacity-100" : "opacity-0")}
|
||||||
|
/>
|
||||||
|
All Creators
|
||||||
|
</CommandItem>
|
||||||
|
{creators.map((creator) => (
|
||||||
|
<CommandItem
|
||||||
|
key={creator}
|
||||||
|
value={creator}
|
||||||
|
onSelect={() => {
|
||||||
|
onChange(creator);
|
||||||
|
setOpen(false);
|
||||||
|
}}
|
||||||
|
>
|
||||||
|
<Check
|
||||||
|
className={cn(
|
||||||
|
"mr-2 h-4 w-4",
|
||||||
|
value === creator ? "opacity-100" : "opacity-0"
|
||||||
|
)}
|
||||||
|
/>
|
||||||
|
<span className="truncate">{creator}</span>
|
||||||
|
</CommandItem>
|
||||||
|
))}
|
||||||
|
</CommandGroup>
|
||||||
|
</CommandList>
|
||||||
|
</Command>
|
||||||
|
</PopoverContent>
|
||||||
|
</Popover>
|
||||||
|
);
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 3: Wire the query through the page**
|
||||||
|
|
||||||
|
In `src/app/(app)/stls/page.tsx`:
|
||||||
|
- Add `getAllPackageCreators` to the import from `@/lib/telegram/queries` (line 3).
|
||||||
|
- Add `getAllPackageCreators()` to the `Promise.all` (line 27) and destructure
|
||||||
|
`availableCreators`:
|
||||||
|
```ts
|
||||||
|
const [result, ingestionStatus, availableTags, availableCreators, skippedCount, ungroupedCount] =
|
||||||
|
await Promise.all([
|
||||||
|
// ...existing entries unchanged...
|
||||||
|
getIngestionStatus(),
|
||||||
|
getAllPackageTags(),
|
||||||
|
getAllPackageCreators(),
|
||||||
|
countSkippedPackages(),
|
||||||
|
countUngroupedPackages(),
|
||||||
|
]);
|
||||||
|
```
|
||||||
|
(Insert `getAllPackageCreators()` immediately after `getAllPackageTags()`, and add
|
||||||
|
`availableCreators` in the matching position of the destructure.)
|
||||||
|
- Pass the prop to `<StlTable>`: `availableCreators={availableCreators}` (next to
|
||||||
|
`availableTags={availableTags}`).
|
||||||
|
|
||||||
|
- [ ] **Step 4: Add prop + handler + render in the table**
|
||||||
|
|
||||||
|
In `src/app/(app)/stls/_components/stl-table.tsx`:
|
||||||
|
- Import the component: `import { CreatorFilter } from "./creator-filter";`
|
||||||
|
- Add to `StlTableProps` (next to `availableTags: string[];`): `availableCreators: string[];`
|
||||||
|
- Add to the destructured params (next to `availableTags,`): `availableCreators,`
|
||||||
|
- Add the handler after `updateTagFilter` (ends ~line 210):
|
||||||
|
```ts
|
||||||
|
const updateCreatorFilter = useCallback(
|
||||||
|
(value: string) => {
|
||||||
|
const params = new URLSearchParams(searchParams.toString());
|
||||||
|
if (value) {
|
||||||
|
params.set("creator", value);
|
||||||
|
params.set("page", "1");
|
||||||
|
} else {
|
||||||
|
params.delete("creator");
|
||||||
|
}
|
||||||
|
router.push(`${pathname}?${params.toString()}`, { scroll: false });
|
||||||
|
},
|
||||||
|
[router, pathname, searchParams]
|
||||||
|
);
|
||||||
|
```
|
||||||
|
- Render it in the toolbar right after the Tags `Select` block (the
|
||||||
|
`{availableTags.length > 0 && (...)}` block, ~lines 507-522):
|
||||||
|
```tsx
|
||||||
|
{availableCreators.length > 0 && (
|
||||||
|
<CreatorFilter
|
||||||
|
creators={availableCreators}
|
||||||
|
value={activeCreator}
|
||||||
|
onChange={updateCreatorFilter}
|
||||||
|
/>
|
||||||
|
)}
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 5: Verify + manual test**
|
||||||
|
|
||||||
|
Run: `npx tsc --noEmit` (filter for the touched files — expect none new). Note: full
|
||||||
|
`npm run build` fails on a pre-existing stale Prisma client issue in
|
||||||
|
`src/lib/telegram/*` unrelated to this change; do not attempt to fix it.
|
||||||
|
Manual: open `/stls`, use the "All Creators" combobox, type to filter, pick a
|
||||||
|
creator → list filters and the "Send all from [Creator]" button appears; pick
|
||||||
|
"All Creators" → filter clears and button disappears.
|
||||||
|
|
||||||
|
- [ ] **Step 6: Commit**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
git add "src/lib/telegram/queries.ts" "src/app/(app)/stls/page.tsx" \
|
||||||
|
"src/app/(app)/stls/_components/creator-filter.tsx" \
|
||||||
|
"src/app/(app)/stls/_components/stl-table.tsx"
|
||||||
|
git commit -m "feat(stls): add searchable creator filter combobox to toolbar"
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Self-Review
|
||||||
|
|
||||||
|
**Spec coverage:**
|
||||||
|
- Req 1 (button when filtered by creator) → Task 2 Step 5 (`{activeCreator && ...}`).
|
||||||
|
- Req 2 (hidden when no filter) → Task 2 Step 5 conditional.
|
||||||
|
- Req 3 (confirm dialog) → Task 2 Step 4 `confirm(...)`.
|
||||||
|
- Req 4 (all pages, not current page) → Task 1 queries `prisma.package.findMany` by `creator`, unpaginated.
|
||||||
|
- Req 5 (skip not-uploaded + dedup live requests) → Task 1 `where` filter on `destChannelId`/`destMessageId` + `existing` check.
|
||||||
|
- Req 6 (report counts) → Task 1 returns `{ queued, skipped }`; Task 2 Step 4 toast.
|
||||||
|
- Req 7 (no polling) → Task 2 handler queues + `router.refresh()`, no poll loop.
|
||||||
|
- Blank-creator guard → Task 1 `creator.trim()` check.
|
||||||
|
|
||||||
|
**Placeholder scan:** No TBD/TODO/placeholder steps; all code shown in full.
|
||||||
|
|
||||||
|
**Type consistency:** Action returns `ActionResult<{ queued: number; skipped: number }>`; handler destructures `result.data.{queued,skipped}` inside the `result.success` branch (where `data` is typed). `sendAllFromCreatorAction` name matches between Task 1 definition, Task 2 import, and Task 2 call site. `activeCreator` defined once (Step 3) and used in Steps 4-5.
|
||||||
@@ -0,0 +1,555 @@
|
|||||||
|
# NAS Backup for Postgres + TDLib State Implementation Plan
|
||||||
|
|
||||||
|
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
||||||
|
|
||||||
|
**Goal:** Add a `backup` container to the DragonsStash stack that takes daily, encrypted, deduplicated backups of the Postgres database and both TDLib state volumes, and ships them to a Synology NAS over NFS.
|
||||||
|
|
||||||
|
**Architecture:** A small Alpine-based image (restic + postgresql16-client + curl + dcron) runs as its own compose service. A crontab fires `backup.sh` daily at 03:00, which dumps Postgres, tars the TDLib volumes, hands both to `restic backup` against an NFS-backed Docker volume, prunes with `restic forget --keep-daily 14`, and reports success/failure to an Uptime Kuma push monitor. Matches the existing `worker`/`bot` pattern: build context in the repo's `docker-compose.yml`, prebuilt image in `/opt/stacks/DragonsStash/docker-compose.yml`, built and pushed by `.drone.yml`.
|
||||||
|
|
||||||
|
**Tech Stack:** Alpine 3.20, restic 0.16, postgresql16-client, dcron, bash, Docker Compose NFS volume driver.
|
||||||
|
|
||||||
|
## Global Constraints
|
||||||
|
|
||||||
|
- Retention: `restic forget --keep-daily 14 --prune` (14-day window, per approved spec).
|
||||||
|
- Schedule: daily backup at 03:00, weekly `restic check` at 04:00 Sunday.
|
||||||
|
- No Docker socket mount, no `privileged: true` — the backup container must not be able to control sibling containers.
|
||||||
|
- No host-level mount — NFS access only via Docker's native `driver_opts: type: nfs` volume, never `/etc/fstab`.
|
||||||
|
- Encryption and retention are restic's job — no hand-rolled `age`/`gpg`/`find -mtime` logic.
|
||||||
|
- TDLib volumes are tarred live (best-effort) — never pause `worker`/`bot` for the backup.
|
||||||
|
- Restore is a manual, documented procedure only — never scripted/automated.
|
||||||
|
|
||||||
|
**Required user input before Task 5 can run:** `NAS_HOST` and `NAS_EXPORT_PATH` (the Synology NFS share details) and a Kuma Push-monitor URL (`KUMA_PUSH_URL`, created manually in the existing Uptime Kuma instance, ~26h expected heartbeat interval). Tasks 1–4 need none of these and can proceed immediately; do not substitute placeholder values for them in Task 5 — stop and ask the user instead.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Task 1: Backup image (Dockerfile + entrypoint)
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Create: `backup/Dockerfile`
|
||||||
|
- Create: `backup/entrypoint.sh`
|
||||||
|
- Create: `backup/crontab`
|
||||||
|
|
||||||
|
**Interfaces:**
|
||||||
|
- Produces: a buildable image tagged `dragonsstash-backup:test` locally, with `/entrypoint.sh` as `ENTRYPOINT`, `/backup.sh` present at the image root (written in Task 2 — this task only needs the `COPY` line and a placeholder-free stub isn't acceptable, so create an empty‑body-but-real `backup/backup.sh` here containing just `#!/bin/bash` + `exit 0`, and Task 2 replaces its contents), `restic`, `pg_dump`/`pg_restore`, `curl`, `tar`, `bash`, `dcron` all on `PATH`.
|
||||||
|
- Consumes: nothing from earlier tasks.
|
||||||
|
|
||||||
|
- [ ] **Step 1: Write `backup/backup.sh` stub**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
#!/bin/bash
|
||||||
|
set -euo pipefail
|
||||||
|
exit 0
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 2: Write `backup/entrypoint.sh`**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
#!/bin/bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
if ! restic snapshots >/dev/null 2>&1; then
|
||||||
|
restic init
|
||||||
|
fi
|
||||||
|
|
||||||
|
exec crond -f -l 2
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 3: Write `backup/crontab`**
|
||||||
|
|
||||||
|
```
|
||||||
|
0 3 * * * /backup.sh >> /proc/1/fd/1 2>&1
|
||||||
|
0 4 * * 0 restic check >> /proc/1/fd/1 2>&1
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 4: Write `backup/Dockerfile`**
|
||||||
|
|
||||||
|
```dockerfile
|
||||||
|
FROM alpine:3.20
|
||||||
|
|
||||||
|
RUN apk add --no-cache restic postgresql16-client curl tzdata dcron tar bash
|
||||||
|
|
||||||
|
COPY backup/backup.sh /backup.sh
|
||||||
|
COPY backup/entrypoint.sh /entrypoint.sh
|
||||||
|
COPY backup/crontab /etc/crontabs/root
|
||||||
|
|
||||||
|
RUN chmod +x /backup.sh /entrypoint.sh
|
||||||
|
|
||||||
|
ENTRYPOINT ["/entrypoint.sh"]
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 5: Build the image**
|
||||||
|
|
||||||
|
Run: `cd /home/sam/Documents/DragonsStash && docker build -t dragonsstash-backup:test -f backup/Dockerfile .`
|
||||||
|
Expected: build completes with `Successfully tagged dragonsstash-backup:test` (or Buildkit's equivalent final `naming to docker.io/library/dragonsstash-backup:test done`), no errors.
|
||||||
|
|
||||||
|
- [ ] **Step 6: Verify the tools are present**
|
||||||
|
|
||||||
|
Run: `docker run --rm dragonsstash-backup:test restic version && docker run --rm dragonsstash-backup:test pg_dump --version`
|
||||||
|
Expected: `restic 0.16.x ...` and `pg_dump (PostgreSQL) 16.x` printed, both commands exit 0.
|
||||||
|
|
||||||
|
- [ ] **Step 7: Verify the entrypoint initializes an empty repo and starts cron**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
mkdir -p /tmp/backup-repo-smoke
|
||||||
|
docker run -d --name backup-smoke \
|
||||||
|
-e RESTIC_REPOSITORY=/backups/restic-repo -e RESTIC_PASSWORD=smoketest \
|
||||||
|
-v /tmp/backup-repo-smoke:/backups \
|
||||||
|
dragonsstash-backup:test
|
||||||
|
sleep 2
|
||||||
|
docker logs backup-smoke
|
||||||
|
docker exec backup-smoke restic snapshots
|
||||||
|
docker rm -f backup-smoke
|
||||||
|
rm -rf /tmp/backup-repo-smoke
|
||||||
|
```
|
||||||
|
|
||||||
|
Expected: `docker logs` shows no errors (restic init ran silently); `restic snapshots` prints an empty snapshot list (repo exists, header row only, no error).
|
||||||
|
|
||||||
|
- [ ] **Step 8: Commit**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
git add backup/Dockerfile backup/entrypoint.sh backup/backup.sh backup/crontab
|
||||||
|
git commit -m "Add backup service image (Dockerfile, entrypoint, crontab)"
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Task 2: `backup.sh` script
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `backup/backup.sh` (replace Task 1's stub with the real script)
|
||||||
|
|
||||||
|
**Interfaces:**
|
||||||
|
- Consumes: the image built in Task 1 (`dragonsstash-backup:test`), rebuilt after this change.
|
||||||
|
- Produces: `/backup.sh`, invoked by cron in Task 1's `crontab` and manually in Task 6's verification. Reads env vars `POSTGRES_USER`, `PGPASSWORD`, `POSTGRES_DB`, `RESTIC_REPOSITORY`, `RESTIC_PASSWORD`, `KUMA_PUSH_URL`. Assumes network hostname `dragonsstash-db:5432` for Postgres and mounts `/data/tdlib-worker`, `/data/tdlib-bot` (read-only) for TDLib state.
|
||||||
|
|
||||||
|
- [ ] **Step 1: Replace `backup/backup.sh` with the real script**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
#!/bin/bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
report_failure() {
|
||||||
|
curl -fsS "$KUMA_PUSH_URL" --get \
|
||||||
|
--data-urlencode "status=down" \
|
||||||
|
--data-urlencode "msg=$BASH_COMMAND failed" || true
|
||||||
|
}
|
||||||
|
trap report_failure ERR
|
||||||
|
|
||||||
|
DUMP_FILE=/tmp/dragonsstash.dump
|
||||||
|
TAR_FILE=/tmp/tdlib.tar.gz
|
||||||
|
|
||||||
|
pg_dump -h dragonsstash-db -U "$POSTGRES_USER" -d "$POSTGRES_DB" -Fc -f "$DUMP_FILE"
|
||||||
|
|
||||||
|
tar czf "$TAR_FILE" -C /data tdlib-worker tdlib-bot
|
||||||
|
|
||||||
|
restic backup "$DUMP_FILE" "$TAR_FILE"
|
||||||
|
restic forget --keep-daily 14 --prune
|
||||||
|
|
||||||
|
rm -f "$DUMP_FILE" "$TAR_FILE"
|
||||||
|
|
||||||
|
curl -fsS "$KUMA_PUSH_URL" --get \
|
||||||
|
--data-urlencode "status=up" \
|
||||||
|
--data-urlencode "msg=OK"
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 2: Rebuild the image**
|
||||||
|
|
||||||
|
Run: `docker build -t dragonsstash-backup:test -f backup/Dockerfile .`
|
||||||
|
Expected: build succeeds.
|
||||||
|
|
||||||
|
- [ ] **Step 3: Stand up a scratch Postgres aliased as `dragonsstash-db`**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker network create backup-test-net 2>/dev/null || true
|
||||||
|
docker run -d --name test-pg --network backup-test-net --network-alias dragonsstash-db \
|
||||||
|
-e POSTGRES_USER=dragons -e POSTGRES_PASSWORD=stash -e POSTGRES_DB=dragonsstash \
|
||||||
|
postgres:16-alpine
|
||||||
|
sleep 5
|
||||||
|
docker exec test-pg pg_isready -U dragons -d dragonsstash
|
||||||
|
```
|
||||||
|
|
||||||
|
Expected: `accepting connections`.
|
||||||
|
|
||||||
|
- [ ] **Step 4: Stand up a mock Kuma push endpoint**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
mkdir -p /tmp/mock-kuma-root && touch /tmp/mock-kuma-root/push
|
||||||
|
docker run -d --name mock-kuma --network backup-test-net \
|
||||||
|
-v /tmp/mock-kuma-root:/srv -w /srv python:3-alpine \
|
||||||
|
python3 -m http.server 8000
|
||||||
|
sleep 1
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 5: Prepare fake TDLib state and a local restic repo dir**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
mkdir -p /tmp/backup-test/tdlib-worker /tmp/backup-test/tdlib-bot /tmp/backup-test/repo
|
||||||
|
echo "fake-session" > /tmp/backup-test/tdlib-worker/state.bin
|
||||||
|
echo "fake-session" > /tmp/backup-test/tdlib-bot/state.bin
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 6: Initialize the test restic repo and run `backup.sh` (success path)**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker run --rm --network backup-test-net \
|
||||||
|
-e RESTIC_REPOSITORY=/backups/restic-repo -e RESTIC_PASSWORD=testpassword \
|
||||||
|
-v /tmp/backup-test/repo:/backups \
|
||||||
|
--entrypoint restic dragonsstash-backup:test init
|
||||||
|
|
||||||
|
docker run --rm --network backup-test-net \
|
||||||
|
-e POSTGRES_USER=dragons -e POSTGRES_PASSWORD=stash -e PGPASSWORD=stash -e POSTGRES_DB=dragonsstash \
|
||||||
|
-e RESTIC_REPOSITORY=/backups/restic-repo -e RESTIC_PASSWORD=testpassword \
|
||||||
|
-e KUMA_PUSH_URL=http://mock-kuma:8000/push \
|
||||||
|
-v /tmp/backup-test/tdlib-worker:/data/tdlib-worker:ro \
|
||||||
|
-v /tmp/backup-test/tdlib-bot:/data/tdlib-bot:ro \
|
||||||
|
-v /tmp/backup-test/repo:/backups \
|
||||||
|
--entrypoint /backup.sh dragonsstash-backup:test
|
||||||
|
echo "exit code: $?"
|
||||||
|
```
|
||||||
|
|
||||||
|
Expected: exit code `0`, restic prints a line like `snapshot xxxxxxxx saved`, no error output.
|
||||||
|
|
||||||
|
- [ ] **Step 7: Verify the snapshot landed and contains both files**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker run --rm -v /tmp/backup-test/repo:/backups \
|
||||||
|
-e RESTIC_REPOSITORY=/backups/restic-repo -e RESTIC_PASSWORD=testpassword \
|
||||||
|
--entrypoint restic dragonsstash-backup:test snapshots
|
||||||
|
|
||||||
|
docker run --rm -v /tmp/backup-test/repo:/backups \
|
||||||
|
-e RESTIC_REPOSITORY=/backups/restic-repo -e RESTIC_PASSWORD=testpassword \
|
||||||
|
--entrypoint restic dragonsstash-backup:test ls latest
|
||||||
|
```
|
||||||
|
|
||||||
|
Expected: `snapshots` shows exactly one entry; `ls latest` lists `/tmp/dragonsstash.dump` and `/tmp/tdlib.tar.gz`.
|
||||||
|
|
||||||
|
- [ ] **Step 8: Verify the failure path reports to Kuma**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker run --rm --network backup-test-net \
|
||||||
|
-e POSTGRES_USER=dragons -e POSTGRES_PASSWORD=wrongpass -e PGPASSWORD=wrongpass -e POSTGRES_DB=dragonsstash \
|
||||||
|
-e RESTIC_REPOSITORY=/backups/restic-repo -e RESTIC_PASSWORD=testpassword \
|
||||||
|
-e KUMA_PUSH_URL=http://mock-kuma:8000/push \
|
||||||
|
-v /tmp/backup-test/tdlib-worker:/data/tdlib-worker:ro \
|
||||||
|
-v /tmp/backup-test/tdlib-bot:/data/tdlib-bot:ro \
|
||||||
|
-v /tmp/backup-test/repo:/backups \
|
||||||
|
--entrypoint /backup.sh dragonsstash-backup:test
|
||||||
|
echo "exit code: $?"
|
||||||
|
docker logs mock-kuma | tail -5
|
||||||
|
```
|
||||||
|
|
||||||
|
Expected: exit code nonzero (pg_dump auth failure trips `set -e`); `docker logs mock-kuma` shows a GET request line containing `status=down`.
|
||||||
|
|
||||||
|
- [ ] **Step 9: Clean up test resources**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker rm -f test-pg mock-kuma
|
||||||
|
docker network rm backup-test-net
|
||||||
|
rm -rf /tmp/backup-test /tmp/mock-kuma-root
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 10: Commit**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
git add backup/backup.sh
|
||||||
|
git commit -m "Implement backup.sh: pg_dump + tdlib tar + restic backup/forget + Kuma reporting"
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Task 3: Wire the `backup` service into the repo's `docker-compose.yml`
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `docker-compose.yml`
|
||||||
|
- Modify: `.env.example`
|
||||||
|
|
||||||
|
**Interfaces:**
|
||||||
|
- Consumes: `backup/Dockerfile` (Task 1), `backup/backup.sh` (Task 2).
|
||||||
|
- Produces: a `backup` compose service buildable via `docker compose build backup`, and a `nas_backups` named volume other tasks (5) will mirror into the production compose file.
|
||||||
|
|
||||||
|
- [ ] **Step 1: Add the `backup` service to `docker-compose.yml`**
|
||||||
|
|
||||||
|
Insert after the existing `bot` service (before `db`):
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
backup:
|
||||||
|
build:
|
||||||
|
context: .
|
||||||
|
dockerfile: backup/Dockerfile
|
||||||
|
pull_policy: never
|
||||||
|
environment:
|
||||||
|
- POSTGRES_USER=${POSTGRES_USER:-dragons}
|
||||||
|
- POSTGRES_PASSWORD=${POSTGRES_PASSWORD:-stash}
|
||||||
|
- PGPASSWORD=${POSTGRES_PASSWORD:-stash}
|
||||||
|
- POSTGRES_DB=${POSTGRES_DB:-dragonsstash}
|
||||||
|
- RESTIC_REPOSITORY=/backups/restic-repo
|
||||||
|
- RESTIC_PASSWORD=${RESTIC_PASSWORD:?Set RESTIC_PASSWORD in .env}
|
||||||
|
- KUMA_PUSH_URL=${KUMA_PUSH_URL:?Set KUMA_PUSH_URL in .env}
|
||||||
|
- TZ=${TZ:-Etc/UTC}
|
||||||
|
volumes:
|
||||||
|
- tdlib_state:/data/tdlib-worker:ro
|
||||||
|
- tdlib_bot_state:/data/tdlib-bot:ro
|
||||||
|
- nas_backups:/backups
|
||||||
|
depends_on:
|
||||||
|
db:
|
||||||
|
condition: service_healthy
|
||||||
|
restart: unless-stopped
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
limits:
|
||||||
|
memory: 256M
|
||||||
|
networks:
|
||||||
|
- backend
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 2: Add the `nas_backups` volume to the `volumes:` block**
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
nas_backups:
|
||||||
|
driver_opts:
|
||||||
|
type: nfs
|
||||||
|
o: "addr=${NAS_HOST},rw,nfsvers=4,soft,timeo=100"
|
||||||
|
device: ":${NAS_EXPORT_PATH}"
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 3: Document the new env vars in `.env.example`**
|
||||||
|
|
||||||
|
Append:
|
||||||
|
|
||||||
|
```
|
||||||
|
# Backup (NAS via NFS + restic)
|
||||||
|
NAS_HOST="" # Synology NAS IP or hostname reachable from this host
|
||||||
|
NAS_EXPORT_PATH="" # NFS export path, e.g. /volume1/dragonsstash-backups
|
||||||
|
RESTIC_PASSWORD="" # generate with: openssl rand -base64 32
|
||||||
|
KUMA_PUSH_URL="" # Uptime Kuma Push monitor URL (create the monitor first)
|
||||||
|
TZ="Etc/UTC"
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 4: Validate the compose file parses**
|
||||||
|
|
||||||
|
Run (with dummy values so the `:?` guards don't fail parsing — `AUTH_SECRET` is required by the existing `app` service, not by this change, but `config` validates the whole file):
|
||||||
|
```bash
|
||||||
|
RESTIC_PASSWORD=dummy KUMA_PUSH_URL=http://dummy NAS_HOST=dummy NAS_EXPORT_PATH=/dummy AUTH_SECRET=dummy \
|
||||||
|
docker compose -f docker-compose.yml config --quiet
|
||||||
|
```
|
||||||
|
Expected: no output, exit code 0 (a syntax/interpolation error would print to stderr and exit nonzero).
|
||||||
|
|
||||||
|
- [ ] **Step 5: Validate the service actually builds through Compose**
|
||||||
|
|
||||||
|
Run:
|
||||||
|
```bash
|
||||||
|
RESTIC_PASSWORD=dummy KUMA_PUSH_URL=http://dummy NAS_HOST=dummy NAS_EXPORT_PATH=/dummy AUTH_SECRET=dummy \
|
||||||
|
docker compose -f docker-compose.yml build backup
|
||||||
|
```
|
||||||
|
Expected: build succeeds (reuses Task 1's image layers).
|
||||||
|
|
||||||
|
- [ ] **Step 6: Commit**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
git add docker-compose.yml .env.example
|
||||||
|
git commit -m "Add backup service and nas_backups volume to docker-compose.yml"
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Task 4: CI — build and push the backup image
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `.drone.yml`
|
||||||
|
|
||||||
|
**Interfaces:**
|
||||||
|
- Consumes: `backup/Dockerfile` (Task 1).
|
||||||
|
- Produces: `git.samagsteribbe.nl/admin/dragonsstash-backup:latest` (and `:<short-sha>`), pushed on every push to `main`. Task 5's production compose file references this image tag.
|
||||||
|
|
||||||
|
- [ ] **Step 1: Add a `build-backup` step, mirroring `build-worker`/`build-bot`**
|
||||||
|
|
||||||
|
Insert after the existing `build-bot` step in `.drone.yml`:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
- name: build-backup
|
||||||
|
image: plugins/docker
|
||||||
|
depends_on: [clone]
|
||||||
|
settings:
|
||||||
|
repo: git.samagsteribbe.nl/admin/dragonsstash-backup
|
||||||
|
registry: git.samagsteribbe.nl
|
||||||
|
dockerfile: backup/Dockerfile
|
||||||
|
tags:
|
||||||
|
- latest
|
||||||
|
- "${DRONE_COMMIT_SHA:0:8}"
|
||||||
|
username:
|
||||||
|
from_secret: gitea_username
|
||||||
|
password:
|
||||||
|
from_secret: gitea_password
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 2: Add `build-backup` to the `deploy` step's `depends_on`**
|
||||||
|
|
||||||
|
Change:
|
||||||
|
```yaml
|
||||||
|
- name: deploy
|
||||||
|
image: alpine
|
||||||
|
depends_on: [build-app, build-worker, build-bot]
|
||||||
|
```
|
||||||
|
to:
|
||||||
|
```yaml
|
||||||
|
- name: deploy
|
||||||
|
image: alpine
|
||||||
|
depends_on: [build-app, build-worker, build-bot, build-backup]
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 3: Validate YAML syntax**
|
||||||
|
|
||||||
|
Run: `python3 -c "import yaml; yaml.safe_load(open('.drone.yml')); print('OK')"`
|
||||||
|
Expected: `OK` printed, no exception.
|
||||||
|
|
||||||
|
- [ ] **Step 4: Commit**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
git add .drone.yml
|
||||||
|
git commit -m "Add build-backup CI step, include it in deploy dependencies"
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Task 5: Wire production (`/opt/stacks/DragonsStash`) — requires NAS + Kuma details from the user
|
||||||
|
|
||||||
|
**Do not start this task until the user has supplied `NAS_HOST` and `NAS_EXPORT_PATH` for the Synology NFS share, and has created an Uptime Kuma Push monitor (name it `dragonsstash-backup`, ~26h expected heartbeat interval) and shared its push URL. If any of these are missing, stop and ask — do not substitute placeholder values here, since this file drives the real deployment.**
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `/opt/stacks/DragonsStash/docker-compose.yml`
|
||||||
|
- Modify: `/opt/stacks/DragonsStash/.env`
|
||||||
|
|
||||||
|
**Interfaces:**
|
||||||
|
- Consumes: `git.samagsteribbe.nl/admin/dragonsstash-backup:latest` (published by Task 4's CI step once merged/pushed), the real `NAS_HOST`/`NAS_EXPORT_PATH`/`KUMA_PUSH_URL` values gathered above.
|
||||||
|
- Produces: a running `dragonsstash-backup` container on the production host, verified in Task 6.
|
||||||
|
|
||||||
|
- [ ] **Step 1: Generate `RESTIC_PASSWORD` and add all four new vars to `/opt/stacks/DragonsStash/.env`**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cd /opt/stacks/DragonsStash
|
||||||
|
printf '\n# Backup (NAS via NFS + restic)\nNAS_HOST="<value from user>"\nNAS_EXPORT_PATH="<value from user>"\nRESTIC_PASSWORD="%s"\nKUMA_PUSH_URL="<value from user>"\nTZ="Etc/UTC"\n' "$(openssl rand -base64 32)" >> .env
|
||||||
|
```
|
||||||
|
|
||||||
|
Replace the two `<value from user>` placeholders with the real NAS details and Kuma push URL before saving — this step cannot be completed with the literal placeholder text left in place.
|
||||||
|
|
||||||
|
- [ ] **Step 2: Add the `backup` service to `/opt/stacks/DragonsStash/docker-compose.yml`**
|
||||||
|
|
||||||
|
Insert after the existing `bot` service (before `db`):
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
backup:
|
||||||
|
image: git.samagsteribbe.nl/admin/dragonsstash-backup:latest
|
||||||
|
container_name: dragonsstash-backup
|
||||||
|
restart: unless-stopped
|
||||||
|
environment:
|
||||||
|
- POSTGRES_USER=${POSTGRES_USER:-dragons}
|
||||||
|
- POSTGRES_PASSWORD=${POSTGRES_PASSWORD:-stash}
|
||||||
|
- PGPASSWORD=${POSTGRES_PASSWORD:-stash}
|
||||||
|
- POSTGRES_DB=${POSTGRES_DB:-dragonsstash}
|
||||||
|
- RESTIC_REPOSITORY=/backups/restic-repo
|
||||||
|
- RESTIC_PASSWORD=${RESTIC_PASSWORD:?Set RESTIC_PASSWORD in .env}
|
||||||
|
- KUMA_PUSH_URL=${KUMA_PUSH_URL:?Set KUMA_PUSH_URL in .env}
|
||||||
|
- TZ=${TZ:-Etc/UTC}
|
||||||
|
volumes:
|
||||||
|
- tdlib_state:/data/tdlib-worker:ro
|
||||||
|
- tdlib_bot_state:/data/tdlib-bot:ro
|
||||||
|
- nas_backups:/backups
|
||||||
|
depends_on:
|
||||||
|
db:
|
||||||
|
condition: service_healthy
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
limits:
|
||||||
|
memory: 256M
|
||||||
|
networks:
|
||||||
|
- internal
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 3: Add the `nas_backups` volume**
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
nas_backups:
|
||||||
|
driver_opts:
|
||||||
|
type: nfs
|
||||||
|
o: "addr=${NAS_HOST},rw,nfsvers=4,soft,timeo=100"
|
||||||
|
device: ":${NAS_EXPORT_PATH}"
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 4: Validate the production compose file parses with the real `.env`**
|
||||||
|
|
||||||
|
Run: `cd /opt/stacks/DragonsStash && docker compose config --quiet`
|
||||||
|
Expected: no output, exit code 0.
|
||||||
|
|
||||||
|
- [ ] **Step 5: Commit is not applicable here** — `/opt/stacks/DragonsStash` is a deployed copy, not the git repo (confirm with `git -C /opt/stacks/DragonsStash status` — expect "not a git repository"). Skip committing; Task 6 deploys these file changes directly.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Task 6: Deploy and verify
|
||||||
|
|
||||||
|
**Files:** none (operational task)
|
||||||
|
|
||||||
|
**Interfaces:**
|
||||||
|
- Consumes: everything from Tasks 1–5.
|
||||||
|
- Produces: a running, verified backup on the real NAS, and one completed restore drill.
|
||||||
|
|
||||||
|
- [ ] **Step 1: Confirm with the user before pushing/deploying**
|
||||||
|
|
||||||
|
Pushing to `main` triggers Drone CI to build all four images and deploy to the production host via SSH (`docker compose pull && docker compose up -d`). Confirm the user wants this to happen now before proceeding — this affects the live stack.
|
||||||
|
|
||||||
|
- [ ] **Step 2: Push to `main`**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cd /home/sam/Documents/DragonsStash
|
||||||
|
git push origin main
|
||||||
|
```
|
||||||
|
|
||||||
|
Expected: Drone pipeline runs `build-app`, `build-worker`, `build-bot`, `build-backup`, then `deploy`, all green. Check via the Drone UI or `drone build info admin/DragonsStash <build-number>` if the `drone` CLI is configured.
|
||||||
|
|
||||||
|
- [ ] **Step 3: Confirm the container is up on the production host**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ssh sam@192.168.68.68 "docker ps --filter name=dragonsstash-backup --format '{{.Names}}\t{{.Status}}'"
|
||||||
|
```
|
||||||
|
|
||||||
|
Expected: `dragonsstash-backup Up ...`.
|
||||||
|
|
||||||
|
- [ ] **Step 4: Trigger one manual backup run and confirm a snapshot lands on the NAS**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ssh sam@192.168.68.68 "docker exec dragonsstash-backup /backup.sh"
|
||||||
|
ssh sam@192.168.68.68 "docker exec dragonsstash-backup restic snapshots"
|
||||||
|
```
|
||||||
|
|
||||||
|
Expected: `backup.sh` exits 0; `restic snapshots` lists exactly one entry.
|
||||||
|
|
||||||
|
- [ ] **Step 5: Confirm the Uptime Kuma push monitor shows green**
|
||||||
|
|
||||||
|
Open the Uptime Kuma dashboard and check the `dragonsstash-backup` monitor's status is up with a recent heartbeat.
|
||||||
|
|
||||||
|
- [ ] **Step 6: Restore drill — prove the backup is actually restorable**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ssh sam@192.168.68.68 "docker exec dragonsstash-backup restic restore latest --target /tmp/restore-drill"
|
||||||
|
ssh sam@192.168.68.68 "docker exec dragonsstash-backup ls -la /tmp/restore-drill/tmp"
|
||||||
|
```
|
||||||
|
|
||||||
|
Expected: `/tmp/restore-drill/tmp/dragonsstash.dump` and `/tmp/restore-drill/tmp/tdlib.tar.gz` both present with nonzero size. Then, on a scratch Postgres (not the live `dragonsstash-db`), confirm the dump restores cleanly:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ssh sam@192.168.68.68 "docker run -d --rm --name restore-drill-pg --network dragonsstash_internal \
|
||||||
|
-e POSTGRES_USER=drill -e POSTGRES_PASSWORD=drill -e POSTGRES_DB=drill postgres:16-alpine"
|
||||||
|
ssh sam@192.168.68.68 "docker cp dragonsstash-backup:/tmp/restore-drill/tmp/dragonsstash.dump /tmp/dragonsstash.dump"
|
||||||
|
ssh sam@192.168.68.68 "docker cp /tmp/dragonsstash.dump restore-drill-pg:/tmp/dragonsstash.dump"
|
||||||
|
ssh sam@192.168.68.68 "docker exec -e PGPASSWORD=drill restore-drill-pg pg_restore -U drill -d drill --clean --if-exists /tmp/dragonsstash.dump"
|
||||||
|
ssh sam@192.168.68.68 "docker exec -e PGPASSWORD=drill restore-drill-pg psql -U drill -d drill -c '\\dt' | head -20"
|
||||||
|
ssh sam@192.168.68.68 "docker rm -f restore-drill-pg"
|
||||||
|
```
|
||||||
|
|
||||||
|
Expected: `pg_restore` completes without fatal errors; `\dt` lists the app's tables (e.g. `Package`, `User`, `TelegramLink`).
|
||||||
|
|
||||||
|
- [ ] **Step 7: Clean up drill artifacts**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ssh sam@192.168.68.68 "docker exec dragonsstash-backup rm -rf /tmp/restore-drill"
|
||||||
|
ssh sam@192.168.68.68 "rm -f /tmp/dragonsstash.dump"
|
||||||
|
```
|
||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,170 @@
|
|||||||
|
# Send All From Creator — Design
|
||||||
|
|
||||||
|
**Date:** 2026-07-04
|
||||||
|
**Status:** Approved for planning
|
||||||
|
|
||||||
|
## Summary
|
||||||
|
|
||||||
|
Add a "Send all from [Creator]" button to the STL packages page that queues every
|
||||||
|
sendable package from a given creator for delivery to the user's Telegram. The
|
||||||
|
button appears in the packages-tab toolbar only when the list is filtered by a
|
||||||
|
single creator (`?creator=<name>`).
|
||||||
|
|
||||||
|
This reuses the existing single-send infrastructure (`BotSendRequest` +
|
||||||
|
`pg_notify('bot_send', id)` + the bot's `send-listener`) and closely mirrors the
|
||||||
|
existing `sendAllInGroupAction`. No bot, DB schema, or send-listener changes are
|
||||||
|
required.
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
Relevant existing pieces:
|
||||||
|
|
||||||
|
- **Creator field:** `Package.creator` is a nullable, indexed `String` on the
|
||||||
|
`Package` model (`prisma/schema.prisma:476-524`). Filtering by exact creator is
|
||||||
|
cheap.
|
||||||
|
- **Creator filter:** The packages list already supports `?creator=<name>`
|
||||||
|
(`src/app/(app)/stls/page.tsx:22,38`), rendered by the packages column as a
|
||||||
|
clickable value (`package-columns.tsx`).
|
||||||
|
- **Single send:** `send-to-telegram-button.tsx` → `POST /api/telegram/bot/send`
|
||||||
|
creates a `BotSendRequest` (status `PENDING`) and fires
|
||||||
|
`pg_notify('bot_send', requestId)`; the button then polls the request to
|
||||||
|
completion.
|
||||||
|
- **Existing bulk send (group):** `sendAllInGroupAction(groupId)` in
|
||||||
|
`src/app/(app)/stls/actions.ts:515-591` fetches the group's packages, filters to
|
||||||
|
those with `destChannelId` + `destMessageId`, skips any with an existing
|
||||||
|
`PENDING`/`SENDING` request for the same package + telegram link, creates a
|
||||||
|
`BotSendRequest` per remaining package, and fires `pg_notify` per package. It is
|
||||||
|
wired into the table via `handleSendAllInGroup` (`stl-table.tsx:264-278`) with a
|
||||||
|
`confirm()` dialog and a success toast — no polling.
|
||||||
|
- **Bot side:** `bot/src/send-listener.ts` processes every `BotSendRequest`
|
||||||
|
uniformly; nothing there needs to change.
|
||||||
|
|
||||||
|
## Requirements
|
||||||
|
|
||||||
|
1. When the packages tab is filtered by a single creator (`?creator=<name>` is
|
||||||
|
present), show a toolbar button labeled **"Send all from [Creator]"**.
|
||||||
|
2. When no creator filter is active, the button is not rendered.
|
||||||
|
3. Clicking the button shows a confirmation dialog before sending (consistent with
|
||||||
|
the single-send and group-send flows, since this pushes files to Telegram).
|
||||||
|
4. On confirm, queue **all** sendable packages from that creator across all pages —
|
||||||
|
not just the current page.
|
||||||
|
5. Skip packages that are not yet uploaded (missing `destChannelId` /
|
||||||
|
`destMessageId`) and packages that already have a `PENDING`/`SENDING` request for
|
||||||
|
the same package + telegram link (dedup).
|
||||||
|
6. Report real counts back to the user via toast, e.g.
|
||||||
|
"Queued 12 packages from [Creator]" and, when applicable, a skipped count.
|
||||||
|
7. No polling — the bulk action queues and returns; sends process in the background
|
||||||
|
via the bot listener (same behavior as group send).
|
||||||
|
|
||||||
|
## Design
|
||||||
|
|
||||||
|
### Server action — `sendAllFromCreatorAction(creatorName: string)`
|
||||||
|
|
||||||
|
New action in `src/app/(app)/stls/actions.ts`, mirroring `sendAllInGroupAction`:
|
||||||
|
|
||||||
|
1. Auth check (`auth()`), return `Unauthorized` if no session.
|
||||||
|
2. Load the user's `TelegramLink`; if none, return
|
||||||
|
"No linked Telegram account. Link one in Settings."
|
||||||
|
3. Reject an empty/blank `creatorName` (guard against sending the entire "no
|
||||||
|
creator" set — see Edge cases).
|
||||||
|
4. Fetch sendable packages directly with a `where` filter (no in-memory filtering):
|
||||||
|
```ts
|
||||||
|
prisma.package.findMany({
|
||||||
|
where: {
|
||||||
|
creator: creatorName,
|
||||||
|
destChannelId: { not: null },
|
||||||
|
destMessageId: { not: null },
|
||||||
|
},
|
||||||
|
select: { id: true },
|
||||||
|
})
|
||||||
|
```
|
||||||
|
5. If none, return "No uploaded packages found for this creator."
|
||||||
|
6. For each package, skip if an existing `PENDING`/`SENDING` `BotSendRequest`
|
||||||
|
exists for that package + telegram link; otherwise create a `BotSendRequest`
|
||||||
|
(status `PENDING`) and fire `pg_notify('bot_send', id)` (best-effort, wrapped in
|
||||||
|
try/catch — the bot also polls).
|
||||||
|
7. `revalidatePath("/stls")`.
|
||||||
|
8. Return counts so the UI can show them. This is a small improvement over
|
||||||
|
`sendAllInGroupAction`, which currently returns `{ success: true, data: undefined }`.
|
||||||
|
Shape:
|
||||||
|
```ts
|
||||||
|
{ success: true, data: { queued: number, skipped: number } }
|
||||||
|
```
|
||||||
|
where `skipped` counts packages that already had a live request. (Packages not
|
||||||
|
yet uploaded are excluded by the query and not counted as skipped.)
|
||||||
|
|
||||||
|
### UI — toolbar button in `stl-table.tsx`
|
||||||
|
|
||||||
|
- Derive the active creator from the URL: `const activeCreator =
|
||||||
|
searchParams.get("creator") ?? "";` (parallel to the existing
|
||||||
|
`activeTag` at `stl-table.tsx:445`).
|
||||||
|
- In the packages-tab toolbar (`stl-table.tsx:478` `flex flex-wrap items-center
|
||||||
|
gap-2` row), render the button only when `activeCreator` is non-empty:
|
||||||
|
```tsx
|
||||||
|
{activeCreator && (
|
||||||
|
<Button variant="outline" size="sm" className="h-9 gap-1.5"
|
||||||
|
onClick={handleSendAllFromCreator}>
|
||||||
|
<Send className="h-3.5 w-3.5" />
|
||||||
|
Send all from {activeCreator}
|
||||||
|
</Button>
|
||||||
|
)}
|
||||||
|
```
|
||||||
|
(Reuse the icon already used by the send action; place near the Upload / Group
|
||||||
|
buttons.)
|
||||||
|
- Add `handleSendAllFromCreator`, modeled on `handleSendAllInGroup`
|
||||||
|
(`stl-table.tsx:264-278`):
|
||||||
|
```ts
|
||||||
|
const handleSendAllFromCreator = useCallback(() => {
|
||||||
|
if (!confirm(`Send all packages from "${activeCreator}" to your Telegram?`)) return;
|
||||||
|
startTransition(async () => {
|
||||||
|
const result = await sendAllFromCreatorAction(activeCreator);
|
||||||
|
if (result.success) {
|
||||||
|
const { queued, skipped } = result.data;
|
||||||
|
toast.success(
|
||||||
|
`Queued ${queued} package${queued === 1 ? "" : "s"} from ${activeCreator}` +
|
||||||
|
(skipped ? ` (${skipped} already queued)` : "")
|
||||||
|
);
|
||||||
|
router.refresh();
|
||||||
|
} else {
|
||||||
|
toast.error(result.error);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}, [activeCreator, router]);
|
||||||
|
```
|
||||||
|
- Import `sendAllFromCreatorAction` alongside the existing `sendAllInGroupAction`
|
||||||
|
import (`stl-table.tsx:51`).
|
||||||
|
|
||||||
|
### Data flow
|
||||||
|
|
||||||
|
```
|
||||||
|
User (filtered by creator) → clicks button → confirm dialog
|
||||||
|
→ sendAllFromCreatorAction(creatorName)
|
||||||
|
→ for each sendable, un-queued package: create BotSendRequest + pg_notify
|
||||||
|
→ returns { queued, skipped } → toast + router.refresh()
|
||||||
|
|
||||||
|
(background) bot send-listener consumes each BotSendRequest → delivers to user
|
||||||
|
```
|
||||||
|
|
||||||
|
## Edge cases
|
||||||
|
|
||||||
|
- **Blank creator:** The action rejects an empty/whitespace `creatorName` so a
|
||||||
|
stray `?creator=` never queues every package. The UI already gates the button on
|
||||||
|
a non-empty `activeCreator`, so this is defense-in-depth.
|
||||||
|
- **No linked Telegram account:** Returns an error toast, queues nothing.
|
||||||
|
- **No uploaded packages for creator:** Returns an error toast, queues nothing.
|
||||||
|
- **All packages already queued:** `queued: 0`, `skipped: N` → toast reflects it.
|
||||||
|
- **pg_notify failure:** Swallowed per-package (best-effort); the bot's periodic
|
||||||
|
poll picks the request up. Matches existing behavior.
|
||||||
|
|
||||||
|
## Out of scope / non-goals
|
||||||
|
|
||||||
|
- No per-row "send all from this creator" action (toolbar-only, per decision).
|
||||||
|
- No progress polling / live completion tracking for the bulk send.
|
||||||
|
- No changes to the bot, `send-listener`, `BotSendRequest` schema, or API routes.
|
||||||
|
- No batching/throttling beyond what the bot listener already does.
|
||||||
|
|
||||||
|
## Files touched
|
||||||
|
|
||||||
|
- `src/app/(app)/stls/actions.ts` — add `sendAllFromCreatorAction`.
|
||||||
|
- `src/app/(app)/stls/_components/stl-table.tsx` — derive `activeCreator`, add
|
||||||
|
`handleSendAllFromCreator`, render the toolbar button, import the new action.
|
||||||
@@ -0,0 +1,205 @@
|
|||||||
|
# NAS Backup for Postgres + TDLib State — Design
|
||||||
|
|
||||||
|
**Date:** 2026-07-23
|
||||||
|
**Status:** Approved for planning
|
||||||
|
|
||||||
|
## Summary
|
||||||
|
|
||||||
|
Add a dedicated `backup` container to the DragonsStash stack that takes daily,
|
||||||
|
encrypted, deduplicated backups of the two things that can't be regenerated —
|
||||||
|
the Postgres database (inventory/STL metadata, users, everything the app
|
||||||
|
manages) and the two TDLib state volumes (Telegram session/auth state for the
|
||||||
|
worker and bot) — and ships them to a Synology NAS over NFS. STL archive
|
||||||
|
contents themselves are explicitly out of scope: they only live on this host
|
||||||
|
temporarily and are not backed up.
|
||||||
|
|
||||||
|
Backups are stored via [restic](https://restic.net/), which provides
|
||||||
|
encryption-at-rest, block-level dedup, and retention pruning natively, so no
|
||||||
|
custom encryption or pruning scripts need to be written or maintained.
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
Current state (as of this design):
|
||||||
|
|
||||||
|
- Production stack runs from `/opt/stacks/DragonsStash/docker-compose.yml` on
|
||||||
|
this Dockge-managed host, pulling prebuilt images from
|
||||||
|
`git.samagsteribbe.nl`. The `docker-compose.yml` in this repo is the
|
||||||
|
build/dev reference and should be kept in sync.
|
||||||
|
- Named volumes in use: `postgres_data` (Postgres 16 data directory),
|
||||||
|
`tdlib_state` (worker's TDLib session), `tdlib_bot_state` (bot's TDLib
|
||||||
|
session), `tmp_zips` and `manual_uploads` (both transient, explicitly out of
|
||||||
|
scope here).
|
||||||
|
- No backup mechanism, NFS mount, or host cron currently exists anywhere in
|
||||||
|
this deployment.
|
||||||
|
- The host already runs Uptime Kuma (used here for backup alerting) and Loki
|
||||||
|
(container logs are presumably already collected there).
|
||||||
|
|
||||||
|
## Requirements
|
||||||
|
|
||||||
|
1. Daily backup of the Postgres database and both TDLib state volumes.
|
||||||
|
2. Backups stored on a Synology NAS via NFS, not on local disk.
|
||||||
|
3. 14-day retention, oldest snapshots pruned automatically.
|
||||||
|
4. Backups encrypted at rest (Postgres dumps and TDLib session files both
|
||||||
|
contain sensitive material — password hashes, Telegram API secrets, live
|
||||||
|
session state).
|
||||||
|
5. Postgres backups must be transactionally consistent regardless of live app
|
||||||
|
traffic. TDLib state backups are best-effort (see Decisions below) — this
|
||||||
|
is an accepted trade-off, not a defect.
|
||||||
|
6. Alert (via existing Uptime Kuma) if a backup run fails or doesn't happen.
|
||||||
|
7. No new host-level state (no `/etc/fstab` entries, no host crontab) — the
|
||||||
|
backup mechanism should be a container, consistent with how everything
|
||||||
|
else on this host is deployed and versioned.
|
||||||
|
8. No new privileged access — specifically, the backup container must not
|
||||||
|
have Docker socket access or any ability to control sibling containers.
|
||||||
|
|
||||||
|
## Decisions
|
||||||
|
|
||||||
|
- **NFS mounted via Docker's native NFS volume driver** (`driver_opts: type:
|
||||||
|
nfs`), not a host-level mount. Keeps all backup-related state inside the
|
||||||
|
compose file instead of split across host config.
|
||||||
|
- **Restic, not hand-rolled tar+age+find.** Restic already solves encryption,
|
||||||
|
dedup, and retention correctly; hand-rolled scripts would be reinventing
|
||||||
|
that logic with more room for bugs.
|
||||||
|
- **TDLib state is tarred live (best-effort), not paused.** Pausing the
|
||||||
|
worker/bot for a clean snapshot would require mounting the Docker socket
|
||||||
|
into the backup container so it could stop/start sibling containers — a
|
||||||
|
real privilege escalation (a compromised backup container could then
|
||||||
|
control any container on the host). The downside of a best-effort tar is
|
||||||
|
bounded: worst case, a bad TDLib restore means redoing the Telegram SMS
|
||||||
|
auth flow, which is the same outcome as having no backup at all. That
|
||||||
|
bounded, low-severity downside doesn't justify the privilege escalation.
|
||||||
|
- **Fixed-time cron (`crond`), not a sleep-loop.** A `sleep 86400` loop drifts
|
||||||
|
on every container restart; a real crontab entry fires at a fixed time of
|
||||||
|
day regardless of restarts, for negligible extra complexity.
|
||||||
|
- **Restore is manual, not automated.** A script capable of restoring can
|
||||||
|
overwrite live state; that should always require a human deliberately
|
||||||
|
running it, not run unattended.
|
||||||
|
|
||||||
|
## Design
|
||||||
|
|
||||||
|
### New service: `backup`
|
||||||
|
|
||||||
|
Added to both `/opt/stacks/DragonsStash/docker-compose.yml` (production) and
|
||||||
|
this repo's `docker-compose.yml` (dev/build reference).
|
||||||
|
|
||||||
|
- **Image**: custom, `FROM alpine:3.20`, `apk add --no-cache restic
|
||||||
|
postgresql16-client curl tzdata dcron tar bash`. No dependency on app
|
||||||
|
source — independent Dockerfile, e.g. `backup/Dockerfile`.
|
||||||
|
- **Scheduling**: `crond -f` in the foreground as the container's entrypoint,
|
||||||
|
with a crontab installed at build time:
|
||||||
|
```
|
||||||
|
0 3 * * * /backup.sh >> /proc/1/fd/1 2>&1
|
||||||
|
0 4 * * 0 restic check >> /proc/1/fd/1 2>&1
|
||||||
|
```
|
||||||
|
(daily dump/backup at 03:00, weekly repo integrity check at 04:00 Sunday).
|
||||||
|
`restic init` runs once at container startup (entrypoint, before `crond`
|
||||||
|
starts), swallowing the "already initialized" error on subsequent
|
||||||
|
container (re)starts.
|
||||||
|
- **Network**: `internal` only — reaches `dragonsstash-db:5432` for
|
||||||
|
`pg_dump`. No ports exposed.
|
||||||
|
- **Volumes**:
|
||||||
|
- `tdlib_state:/data/tdlib-worker:ro`
|
||||||
|
- `tdlib_bot_state:/data/tdlib-bot:ro`
|
||||||
|
- `nas_backups:/backups`, a named volume defined with:
|
||||||
|
```yaml
|
||||||
|
nas_backups:
|
||||||
|
driver_opts:
|
||||||
|
type: nfs
|
||||||
|
o: "addr=${NAS_HOST},rw,nfsvers=4,soft,timeo=100"
|
||||||
|
device: ":${NAS_EXPORT_PATH}"
|
||||||
|
```
|
||||||
|
- **New `.env` entries**: `NAS_HOST`, `NAS_EXPORT_PATH` (NFS share details),
|
||||||
|
`RESTIC_PASSWORD` (repo encryption key), `KUMA_PUSH_URL` (Uptime Kuma push
|
||||||
|
monitor URL). All four are inputs to gather during implementation, not
|
||||||
|
hardcoded.
|
||||||
|
- `restart: unless-stopped`, no `privileged`, no Docker socket mount.
|
||||||
|
|
||||||
|
### `backup.sh`
|
||||||
|
|
||||||
|
```
|
||||||
|
set -euo pipefail
|
||||||
|
trap 'curl -fsS "$KUMA_PUSH_URL" --get --data-urlencode "status=down" \
|
||||||
|
--data-urlencode "msg=$BASH_COMMAND failed"' ERR
|
||||||
|
|
||||||
|
pg_dump -h dragonsstash-db -U "$POSTGRES_USER" -d "$POSTGRES_DB" \
|
||||||
|
-Fc -f /tmp/dragonsstash.dump
|
||||||
|
|
||||||
|
tar czf /tmp/tdlib.tar.gz -C /data tdlib-worker tdlib-bot
|
||||||
|
|
||||||
|
restic backup /tmp/dragonsstash.dump /tmp/tdlib.tar.gz
|
||||||
|
restic forget --keep-daily 14 --prune
|
||||||
|
|
||||||
|
rm -f /tmp/dragonsstash.dump /tmp/tdlib.tar.gz
|
||||||
|
|
||||||
|
curl -fsS "$KUMA_PUSH_URL" --get --data-urlencode "status=up" \
|
||||||
|
--data-urlencode "msg=OK"
|
||||||
|
```
|
||||||
|
|
||||||
|
`PGPASSWORD` and `RESTIC_REPOSITORY=/backups/restic-repo` are set as
|
||||||
|
container environment variables (from `.env`), not inline in the script.
|
||||||
|
|
||||||
|
### Data flow
|
||||||
|
|
||||||
|
```
|
||||||
|
crond (daily 03:00)
|
||||||
|
→ pg_dump (consistent snapshot via Postgres MVCC) → /tmp/dragonsstash.dump
|
||||||
|
→ tar tdlib_state + tdlib_bot_state (best-effort, live) → /tmp/tdlib.tar.gz
|
||||||
|
→ restic backup (encrypt + dedup) → NFS-mounted repo on Synology NAS
|
||||||
|
→ restic forget --keep-daily 14 --prune
|
||||||
|
→ curl Uptime Kuma push monitor (up on success, down + reason on any failure)
|
||||||
|
```
|
||||||
|
|
||||||
|
### Restore (manual, documented procedure — not scripted/automated)
|
||||||
|
|
||||||
|
```
|
||||||
|
restic -r /backups/restic-repo restore latest --target /tmp/restore
|
||||||
|
pg_restore -h dragonsstash-db -U "$POSTGRES_USER" -d "$POSTGRES_DB" \
|
||||||
|
--clean --if-exists /tmp/restore/tmp/dragonsstash.dump
|
||||||
|
# untar /tmp/restore/tmp/tdlib.tar.gz back into the tdlib_state /
|
||||||
|
# tdlib_bot_state volumes (via a throwaway container mounting both)
|
||||||
|
```
|
||||||
|
|
||||||
|
## Alerting
|
||||||
|
|
||||||
|
- One Uptime Kuma **Push** monitor, created manually in the existing Kuma
|
||||||
|
instance, with an expected heartbeat interval of ~26 hours (slack past the
|
||||||
|
24h schedule so one slow run doesn't false-positive). Whatever notification
|
||||||
|
channels are already configured on that monitor fire automatically — no new
|
||||||
|
alerting integration.
|
||||||
|
- `backup.sh` pushes `status=up` on success and `status=down` (with the
|
||||||
|
failing command in `msg`) on any failure, via the `ERR` trap.
|
||||||
|
- Container logs go to stdout, collected the same way every other container's
|
||||||
|
logs already are on this host.
|
||||||
|
|
||||||
|
## Testing
|
||||||
|
|
||||||
|
This repo has no automated test framework (documented convention: manual
|
||||||
|
testing). For this infra change:
|
||||||
|
|
||||||
|
- After deploy: manually run `docker exec dragonsstash-backup /backup.sh`
|
||||||
|
once, confirm a snapshot appears (`restic snapshots`), confirm the Kuma
|
||||||
|
monitor goes green.
|
||||||
|
- **Restore drill** (once, during setup): actually restore the dump into a
|
||||||
|
scratch Postgres and untar the TDLib archive into scratch volumes, to prove
|
||||||
|
the backup is really restorable. Not automated or recurring for now.
|
||||||
|
|
||||||
|
## Out of scope / non-goals
|
||||||
|
|
||||||
|
- Backing up `tmp_zips` or `manual_uploads` — both transient by design.
|
||||||
|
- Automated/scheduled restore testing.
|
||||||
|
- Backing up any other stack on this host (this design is DragonsStash-only,
|
||||||
|
though the pattern — Docker-native NFS volume + restic — could be reused
|
||||||
|
for other stacks later).
|
||||||
|
- Pausing worker/bot for a guaranteed-consistent TDLib snapshot (see
|
||||||
|
Decisions).
|
||||||
|
|
||||||
|
## Files touched
|
||||||
|
|
||||||
|
- `docker-compose.yml` (this repo) and
|
||||||
|
`/opt/stacks/DragonsStash/docker-compose.yml` (production) — add `backup`
|
||||||
|
service, `nas_backups` volume.
|
||||||
|
- `backup/Dockerfile` — new.
|
||||||
|
- `backup/backup.sh` — new.
|
||||||
|
- `backup/crontab` — new.
|
||||||
|
- `.env.example` / `.env` — add `NAS_HOST`, `NAS_EXPORT_PATH`,
|
||||||
|
`RESTIC_PASSWORD`, `KUMA_PUSH_URL`.
|
||||||
@@ -0,0 +1,271 @@
|
|||||||
|
# Provenance Backfill on Re-Index — Design
|
||||||
|
|
||||||
|
**Date:** 2026-07-23
|
||||||
|
**Status:** Approved for planning
|
||||||
|
|
||||||
|
## Summary
|
||||||
|
|
||||||
|
Recover the true origin of packages whose recorded "source" is a placeholder —
|
||||||
|
specifically the manual-upload and `rebuild.ts`-created records whose
|
||||||
|
`sourceChannelId` points at the destination (archive) channel itself. When a
|
||||||
|
real source channel is re-indexed and the worker encounters an archive that
|
||||||
|
matches such a placeholder package, it backfills the real
|
||||||
|
`sourceChannelId` / `sourceMessageId` / `sourceTopicId` / `sourceCaption` /
|
||||||
|
`creator` (and, for records that never had one, the file listing and preview)
|
||||||
|
onto the existing package — **without downloading the full archive**.
|
||||||
|
|
||||||
|
Matching is a two-stage process: a zero-download candidate lookup by
|
||||||
|
`fileName` + total `fileSize`, confirmed by a CRC32 fingerprint read from just
|
||||||
|
the archive's central directory via a **ranged (tail) download** of a few
|
||||||
|
KB–MB, instead of the multi-GB whole file.
|
||||||
|
|
||||||
|
This is the first of four linked sub-projects. The others (creator
|
||||||
|
normalization, provenance display, missing-files) are out of scope here and get
|
||||||
|
their own spec → plan cycles. "Missing files" is explicitly deferred until this
|
||||||
|
lands.
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
Current state (as of this design):
|
||||||
|
|
||||||
|
- `Package` records their origin via `sourceChannelId`, `sourceMessageId`,
|
||||||
|
`sourceTopicId`, `sourceCaption`, and (for content dedup) `contentHash` and
|
||||||
|
`remoteUniqueId`. `creator` is a free-text string extracted at ingestion.
|
||||||
|
- Two ingestion paths create records with **placeholder provenance**, where the
|
||||||
|
"source" is the destination channel, not a real origin:
|
||||||
|
- **Manual uploads** (`worker/src/manual-upload.ts`) set
|
||||||
|
`sourceChannelId = destChannel.id` and `sourceMessageId = destResult.messageId`.
|
||||||
|
They *do* populate `PackageFile` (with `crc32`) from a local central-directory
|
||||||
|
read at upload time.
|
||||||
|
- **`rebuild.ts`** scans the destination channel and creates minimal records
|
||||||
|
with `fileCount == 0` and **no** `PackageFile` rows.
|
||||||
|
- The main worker upload path (`worker/src/worker.ts` → `uploadToChannel`) sends
|
||||||
|
archives to the destination channel with **no caption**, so provenance cannot
|
||||||
|
be recovered by re-reading destination messages — it has to come from matching
|
||||||
|
against real source channels.
|
||||||
|
- The scan/dedup ladder in `processOneArchiveSet` (`worker/src/worker.ts:1521`)
|
||||||
|
is entirely **source-channel-scoped**:
|
||||||
|
1. `findPackageByRemoteUniqueId(channel.id, …)` — same channel only
|
||||||
|
2. `packageExistsBySourceMessage(channel.id, …)` — same channel only
|
||||||
|
3. `findRepostedPackage(channel.id, fileName, size)` — same channel only
|
||||||
|
4. …then full download → `packageExistsByHash(contentHash)` — global, but only
|
||||||
|
*after* downloading the whole archive.
|
||||||
|
- Because a placeholder package's `sourceChannelId` is the destination, checks
|
||||||
|
1–3 never match it when a real source channel is scanned. Today the worker
|
||||||
|
therefore downloads the entire archive, hits check 4, finds it is a duplicate,
|
||||||
|
and skips — **wasting the download and leaving the wrong provenance in place.**
|
||||||
|
- There is already an in-production precedent for "match an already-known
|
||||||
|
archive during scan and enrich its metadata": when `findRepostedPackage`
|
||||||
|
matches, the worker backfills richer *topic* context onto the existing package
|
||||||
|
via `updatePackageTopicContext` (`worker/src/worker.ts:1608`+). Provenance
|
||||||
|
backfill is the same move, widened from same-channel to cross-channel.
|
||||||
|
- `remote.unique_id` is **not** a reliable cross-channel key: it identifies a
|
||||||
|
stored file object on Telegram's servers, not the content. The same archive
|
||||||
|
independently uploaded to two channels gets different `unique_id`s; it only
|
||||||
|
matches for forwarded messages (same underlying file object). This is why the
|
||||||
|
existing dedup scopes it to a single channel.
|
||||||
|
|
||||||
|
## Goals / Non-Goals
|
||||||
|
|
||||||
|
**Goals**
|
||||||
|
|
||||||
|
- Attribute true origin to placeholder-provenance packages during normal
|
||||||
|
re-index scans, opportunistically (no separate pass).
|
||||||
|
- Do it without downloading whole archives (ranged central-directory read only).
|
||||||
|
- Be non-destructive: only ever touch packages that currently have placeholder
|
||||||
|
provenance; never overwrite a real, non-placeholder source.
|
||||||
|
- Be idempotent: re-running a re-index does not re-mutate or duplicate.
|
||||||
|
|
||||||
|
**Non-Goals (separate sub-projects / deferred)**
|
||||||
|
|
||||||
|
- Creator name normalization / canonical creator entity.
|
||||||
|
- UI changes to display provenance.
|
||||||
|
- Recovering "missing files."
|
||||||
|
- Choosing a *preferred* origin when an archive genuinely exists in several
|
||||||
|
source channels (first confirmed source wins).
|
||||||
|
|
||||||
|
## Design
|
||||||
|
|
||||||
|
### 1. Candidate definition
|
||||||
|
|
||||||
|
> A package is a backfill candidate iff **`sourceChannelId == destChannelId`
|
||||||
|
> OR `sourceMessageId == 0`**.
|
||||||
|
|
||||||
|
Two placeholder shapes exist (verified against live data 2026-07-23):
|
||||||
|
- **Manual uploads** (`manual-upload.ts`): `sourceChannelId == destChannelId`,
|
||||||
|
real `contentHash`, real `sourceMessageId`, has a `PackageFile` listing.
|
||||||
|
- **Rebuild records** (`rebuild.ts`): `sourceMessageId == 0n` (deliberate
|
||||||
|
"unknown" sentinel), synthetic `contentHash = "rebuild:<destChannelId>:<destMessageId>"`,
|
||||||
|
`fileCount == 0`, and `sourceChannelId` set to an **arbitrary fallback source
|
||||||
|
channel** (`sourceChannels[0]`) — NOT the destination. (This is the common
|
||||||
|
case: e.g. 59,893 records after a destination rebuild.)
|
||||||
|
|
||||||
|
Normal ingestion always sets a real `sourceMessageId` (> 0) and a real source
|
||||||
|
channel, so neither marker matches a genuinely-sourced package. Both markers are
|
||||||
|
overwritten on backfill (source channel + message become real), so a record
|
||||||
|
stops being a candidate once fixed — this is what makes re-scans idempotent.
|
||||||
|
|
||||||
|
**Known limitation:** backfill does NOT rewrite a rebuild record's synthetic
|
||||||
|
`"rebuild:"` `contentHash` (the true content hash would require a full download,
|
||||||
|
which this feature avoids). That is acceptable — dedup after backfill relies on
|
||||||
|
`remoteUniqueId` + name/size within the source channel, not on `contentHash`.
|
||||||
|
|
||||||
|
### 2. Where it hooks
|
||||||
|
|
||||||
|
A new step is inserted into `processOneArchiveSet` **between check #3
|
||||||
|
(`findRepostedPackage`) and the full download**. It runs only after the
|
||||||
|
same-channel checks have missed (so genuine same-channel reposts keep their
|
||||||
|
existing fast paths).
|
||||||
|
|
||||||
|
Flow for the scanned archive set:
|
||||||
|
|
||||||
|
1. Stage A — **candidate lookup (zero download).** Query for a package where
|
||||||
|
`(sourceChannelId == destChannelId OR sourceMessageId == 0)` AND
|
||||||
|
`fileName == archiveName` AND `fileSize == totalArchiveSize`.
|
||||||
|
(`Package` has `@@index([fileName])`.)
|
||||||
|
- No candidate → fall through to normal ingestion unchanged.
|
||||||
|
2. Stage B — **fingerprint confirmation (tiny download).** See §3.
|
||||||
|
3. On confirmation → **backfill** (see §4) and return `null` (treated as a
|
||||||
|
duplicate; no full download, no new package). Increment a `zipsBackfilled`
|
||||||
|
counter.
|
||||||
|
4. On mismatch / failure / ambiguity → fall through to normal ingestion (see §6).
|
||||||
|
|
||||||
|
### 3. Fingerprint confirmation
|
||||||
|
|
||||||
|
The confirmation signal is the **multiset of internal CRC32s** of the archive's
|
||||||
|
entries (CRC32 of each entry's *uncompressed* data). This value is a property of
|
||||||
|
the archive contents and is identical regardless of which channel hosts the
|
||||||
|
file.
|
||||||
|
|
||||||
|
- **Candidate side:** for manual uploads, `PackageFile.crc32` is already
|
||||||
|
populated from the local central-directory read — zero cost. For **rebuild
|
||||||
|
records** (`fileCount == 0`, no `crc32`), there is nothing stored to compare
|
||||||
|
against; obtain the candidate's fingerprint with a **second ranged tail read
|
||||||
|
of its destination copy** (ZIP/7z — still no full download). RAR candidates
|
||||||
|
cannot be tail-read on either side, so they fall back to name+size (see §5).
|
||||||
|
- **Scanned-source side:** read via a **ranged (tail) download** of the central
|
||||||
|
directory:
|
||||||
|
- **ZIP** (incl. multipart): the End-of-Central-Directory record + central
|
||||||
|
directory live at the tail of the last part. Download only that tail and
|
||||||
|
parse entries. (Reuses the lightweight-listing mechanism that is
|
||||||
|
sub-project 4's core.)
|
||||||
|
- **7z**: a start header at byte 0 points to an end header at the tail; fetch
|
||||||
|
both small pieces.
|
||||||
|
- **RAR**: headers are scattered through the file — no cheap tail read. See §5.
|
||||||
|
- **Match rule:** confirmed iff the sorted CRC32 multisets are equal **and** file
|
||||||
|
counts are equal.
|
||||||
|
|
||||||
|
The comparison logic (CRC32 multiset equality) and the candidate-match predicate
|
||||||
|
are implemented as **pure functions** with no TDLib dependency, so they are unit
|
||||||
|
testable (see §7).
|
||||||
|
|
||||||
|
### 4. What gets written
|
||||||
|
|
||||||
|
On a confirmed match, update the candidate package in a single transaction,
|
||||||
|
overwriting only placeholder/empty fields:
|
||||||
|
|
||||||
|
| Field | New value | Condition |
|
||||||
|
|---|---|---|
|
||||||
|
| `sourceChannelId` | scanned `channel.id` | always |
|
||||||
|
| `sourceMessageId` | scanned `parts[0].id` | always |
|
||||||
|
| `sourceTopicId` | `ctx.sourceTopicId` | always |
|
||||||
|
| `sourceCaption` | scanned message caption | always |
|
||||||
|
| `remoteUniqueId` | scanned `firstRemoteUniqueId` | always (enables future same-channel dedup via check #1) |
|
||||||
|
| `creator` | re-derived from source (topic > filename > channel) | **always** — un-normalized; the later creator-normalization sub-project cleans it up |
|
||||||
|
| `fileCount` + `PackageFile[]` | from the central-directory read | only if candidate had none (`fileCount == 0`) — folds in the sub-project 4 outcome for rebuild records |
|
||||||
|
| `previewData` / `previewMsgId` | matched preview from scan | only if candidate has none |
|
||||||
|
|
||||||
|
**Left untouched:** `contentHash`, `destChannelId`, `destMessageId`,
|
||||||
|
`destMessageIds` — the bytes physically live in the destination channel; that is
|
||||||
|
correct and must not change.
|
||||||
|
|
||||||
|
**Transaction safety:** re-check `sourceChannelId == destChannelId` *inside* the
|
||||||
|
transaction before writing, so a concurrent worker that already backfilled the
|
||||||
|
record causes this one to no-op (mirrors the existing `backfill.ts` guard).
|
||||||
|
|
||||||
|
### 5. RAR handling
|
||||||
|
|
||||||
|
RAR sources cannot be tail-read, so no cheap CRC fingerprint is available.
|
||||||
|
Decision: **backfill RAR matches on `fileName` + `fileSize` alone**, flagged as
|
||||||
|
lower-confidence in logs and via a `SystemNotification`, so they can be audited.
|
||||||
|
No full download.
|
||||||
|
|
||||||
|
This name+size-only fallback applies to any candidate that cannot produce a
|
||||||
|
CRC32 fingerprint cheaply: a RAR **source**, a RAR **candidate**, or a rebuild
|
||||||
|
candidate whose destination copy is RAR. ZIP/7z rebuild candidates are still
|
||||||
|
confirmed by fingerprint via the second tail read described in §3.
|
||||||
|
|
||||||
|
### 6. Conflict, mismatch, ambiguity
|
||||||
|
|
||||||
|
- **Fingerprint mismatch** → the scanned archive is genuinely different content
|
||||||
|
that merely shares name+size. Fall through to **normal ingestion** (download +
|
||||||
|
index) — it is a new package for this source.
|
||||||
|
- **Ambiguous candidates** (2+ match name+size and the fingerprint cannot
|
||||||
|
disambiguate) → log a `SystemNotification`, backfill nothing, and fall through
|
||||||
|
to normal ingestion; the post-download `packageExistsByHash` check still
|
||||||
|
dedups it safely.
|
||||||
|
- **Same archive in multiple source channels** → the **first re-indexed source
|
||||||
|
that confirms wins.** After backfill the package has a real source, so it is no
|
||||||
|
longer a candidate; later scans treat it as an ordinary duplicate via checks #1
|
||||||
|
(the `remoteUniqueId` we set) or #3.
|
||||||
|
|
||||||
|
### 7. Idempotency
|
||||||
|
|
||||||
|
Falls out of the candidate definition. Once backfilled:
|
||||||
|
- `sourceChannelId` is the real channel → no longer a candidate.
|
||||||
|
- A re-scan of that source hits check #1 (`remoteUniqueId`, now set) or check #3
|
||||||
|
(`findRepostedPackage`) → normal dedup skip. No re-mutation, no duplicate.
|
||||||
|
|
||||||
|
### 8. Error handling
|
||||||
|
|
||||||
|
- **Tail-download failure** (network / `FLOOD_WAIT`) → cannot confirm this round.
|
||||||
|
Do **not** fall back to a full download and do **not** guess: leave the
|
||||||
|
candidate untouched and let the next re-index retry. Wrap the ranged read in
|
||||||
|
`withFloodWait` (per the TDLib skill).
|
||||||
|
- All new TDLib calls follow the skill's patterns: `withFloodWait`, listener
|
||||||
|
attached before the async op, client closed in `finally`.
|
||||||
|
|
||||||
|
### 9. Visibility
|
||||||
|
|
||||||
|
- Add a `zipsBackfilled` counter to `IngestionRun` activity so a re-index run
|
||||||
|
reports "N provenance backfills" instead of silently mutating records.
|
||||||
|
- Info log per backfill (candidate id, old vs new source, confidence:
|
||||||
|
fingerprint | name+size-RAR).
|
||||||
|
- `SystemNotification` for ambiguous-candidate cases.
|
||||||
|
|
||||||
|
## Testing
|
||||||
|
|
||||||
|
The repo currently has no test framework (`CLAUDE.md`: "testing is manual"). This
|
||||||
|
work introduces a **lightweight test harness** for the worker (e.g. `vitest` or
|
||||||
|
node's built-in `node:test`) and unit-tests the correctness-critical pure logic:
|
||||||
|
|
||||||
|
- **Unit (automated):**
|
||||||
|
- CRC32 multiset fingerprint equality (equal sets, different order, differing
|
||||||
|
counts, disjoint sets).
|
||||||
|
- Candidate-match predicate (name+size+placeholder true/false cases).
|
||||||
|
- Field-merge rules (which fields overwrite, which only fill-if-empty).
|
||||||
|
- **Manual integration checklist:**
|
||||||
|
1. Re-index a real source channel containing a known manually-uploaded pack →
|
||||||
|
verify `sourceChannelId`/`sourceMessageId`/`sourceCaption`/`creator` are
|
||||||
|
backfilled and no full download occurs.
|
||||||
|
2. A rebuild-created record (`fileCount == 0`) → verify listing + provenance
|
||||||
|
are both populated from the single tail read.
|
||||||
|
3. A RAR pack → verify name+size backfill with the lower-confidence log/notice.
|
||||||
|
4. Re-run the same re-index → verify it is a no-op (idempotent).
|
||||||
|
5. A genuine name+size collision (different content) → verify it ingests as a
|
||||||
|
new package rather than being mis-attributed.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
None blocking. Preferred-origin selection among multiple real sources is
|
||||||
|
deliberately out of scope (first confirmed wins).
|
||||||
|
|
||||||
|
## Affected Code (indicative, for planning)
|
||||||
|
|
||||||
|
- `worker/src/worker.ts` — `processOneArchiveSet`: new Stage A/B step; new counter.
|
||||||
|
- `worker/src/archive/` — ranged central-directory reader (ZIP/7z tail); shared
|
||||||
|
with sub-project 4. Pure CRC32-fingerprint compare helper.
|
||||||
|
- `worker/src/tdlib/download.ts` — ranged/partial download support (`offset`/`limit`).
|
||||||
|
- `worker/src/db/queries.ts` — candidate lookup + transactional backfill update.
|
||||||
|
- `prisma/schema.prisma` — `IngestionRun.zipsBackfilled` (and run-counter plumbing).
|
||||||
|
- Worker test harness + first unit tests.
|
||||||
@@ -0,0 +1,180 @@
|
|||||||
|
# Ranged inner-file listing for RAR & 7z — design
|
||||||
|
|
||||||
|
**Date:** 2026-07-27
|
||||||
|
**Status:** Approved (design), pending spec review → implementation plan
|
||||||
|
|
||||||
|
## Problem
|
||||||
|
|
||||||
|
The reindex/provenance-backfill path (`worker/src/provenance-backfill.ts`) can index an
|
||||||
|
archive's inner files *without* re-downloading it, by reading the file listing from a small
|
||||||
|
ranged read of the copy already in the source/destination channel. This works **only for
|
||||||
|
ZIP** today (ZIP keeps its central directory in a tail that `parseZipCentralDirectoryFromTail`
|
||||||
|
reads). For **RAR and 7z**, `tryProvenanceBackfill` backfills provenance (creator, source
|
||||||
|
channel, `remoteUniqueId`) so the file is skipped on re-scan and never re-downloaded — but it
|
||||||
|
leaves the inner listing empty (`fileCount = 0`), because `scannedEntries` is only computed for
|
||||||
|
ZIP.
|
||||||
|
|
||||||
|
Scope of the gap (rebuild placeholders with `fileCount = 0`, as of 2026-07-27):
|
||||||
|
|
||||||
|
| Type | Count | Total | Avg | Max | Multipart |
|
||||||
|
|---|---|---|---|---|---|
|
||||||
|
| 7z | 19,646 | 14 TB | 0.73 GB | 3.9 GB | 0 |
|
||||||
|
| RAR | 12,780 | 13 TB | 1.04 GB | 116 GB | 1,016 |
|
||||||
|
|
||||||
|
A "just download the whole archive" fallback for all of these means ~27 TB of re-downloads —
|
||||||
|
the exact cost this path exists to avoid.
|
||||||
|
|
||||||
|
## Goals
|
||||||
|
|
||||||
|
- Index the inner files (names + sizes; CRCs where cheaply available) of RAR and 7z
|
||||||
|
placeholders **without** downloading the whole archive in the common case.
|
||||||
|
- Reuse the existing, battle-tested CLI listing parsers (`parse7zOutput`,
|
||||||
|
`parseUnrarTechnical`) rather than reimplementing filename/size/CRC extraction.
|
||||||
|
- Keep cost proportional to **file count**, not archive size (so even the 116 GB RAR is cheap).
|
||||||
|
- Guarantee a listing for the rare archives the cheap path can't handle, via a full-download
|
||||||
|
fallback that respects the existing max-size guard.
|
||||||
|
|
||||||
|
## Non-goals
|
||||||
|
|
||||||
|
- No change to ingestion of genuinely-new files (those are downloaded in full to be re-uploaded
|
||||||
|
regardless, so a cheap listing does not help there). This feature only affects the
|
||||||
|
provenance-backfill / skip path.
|
||||||
|
- No new ZIP behaviour — the existing ZIP tail reader stays as-is.
|
||||||
|
- Not attempting to list password-encrypted-header archives from ranged reads (no password);
|
||||||
|
those take the fallback and, if oversized, are flagged.
|
||||||
|
|
||||||
|
## Approach (chosen: "harvest header regions → sparse file → native CLI")
|
||||||
|
|
||||||
|
Do the *minimum* binary parsing needed to locate an archive's header bytes, fetch only those
|
||||||
|
via ranged reads, write them into a sparse temp file at their true offsets (data regions left
|
||||||
|
as unwritten zero holes → ~no disk use), then run the real `7z l` / `unrar lt` and reuse the
|
||||||
|
existing parsers. The native tools do the hard parsing (7z's LZMA-encoded headers, RAR's two
|
||||||
|
format versions, Unicode names) — we only compute where the headers are.
|
||||||
|
|
||||||
|
Rejected alternatives: full native TS parsers (most custom binary code, highest risk);
|
||||||
|
RAR5-quick-open-only (most community RARs lack it → collapses to ~13 TB of RAR downloads).
|
||||||
|
|
||||||
|
## Components
|
||||||
|
|
||||||
|
New directory `worker/src/archive/ranged/`, one focused module per concern, all returning the
|
||||||
|
existing `FileEntry[]` type from `zip-reader.ts`:
|
||||||
|
|
||||||
|
- `sparse-list.ts` — `listFromSparse(parts, runner, parse) → FileEntry[] | null`, where each
|
||||||
|
`part` is `{ fileName, size, regions: {offset, bytes}[] }`. For each part it writes a sparse
|
||||||
|
temp file (`truncate` to `size`, then write only the header `regions` at their offsets),
|
||||||
|
co-locates all parts in one temp dir under their real names, invokes the supplied CLI runner
|
||||||
|
(`7z l` / `unrar lt`) on the first part, feeds stdout to the supplied `parse` fn
|
||||||
|
(`parse7zOutput` / `parseUnrarTechnical`), and cleans up. Single-part archives are just the
|
||||||
|
one-element case. Returns `null` on CLI error / empty parse.
|
||||||
|
- `sevenz-ranged.ts` — `readSevenZListingRanged(client, parts) → FileEntry[] | null`.
|
||||||
|
- `rar-ranged.ts` — `readRarListingRanged(client, parts) → FileEntry[] | null`.
|
||||||
|
- Dispatcher in `provenance-backfill.ts`: `readScannedListingRanged(archiveType, client, parts)`
|
||||||
|
replacing the current `if (archiveType === "ZIP")` branch; the destination-copy read in
|
||||||
|
`resolveCandidateFingerprintEntries` gets the same dispatch.
|
||||||
|
|
||||||
|
`FileEntry` shape (unchanged): `{ path, fileName, extension, compressedSize, uncompressedSize, crc32 }`.
|
||||||
|
|
||||||
|
### 7z ranged listing
|
||||||
|
|
||||||
|
7z layout: 32-byte signature header at offset 0 → packed streams → end header (lists files) at
|
||||||
|
the end; the signature header stores the end header's location.
|
||||||
|
|
||||||
|
1. Ranged-read `[0, 32)`; validate magic `37 7A BC AF 27 1C`. Read LE `uint64`
|
||||||
|
`NextHeaderOffset` (byte 12) and `NextHeaderSize` (byte 20). End header is at absolute offset
|
||||||
|
`32 + NextHeaderOffset`, length `NextHeaderSize`.
|
||||||
|
2. Ranged-read `[32 + NextHeaderOffset, NextHeaderSize)` — the "next header".
|
||||||
|
3. Branch on the next header's first byte (a 7z property id):
|
||||||
|
- **`0x01` (kHeader, plain/uncompressed header):** two regions suffice —
|
||||||
|
`{0: sigHeader}` and `{32+NextHeaderOffset: endHeader}`.
|
||||||
|
- **`0x17` (kEncodedHeader, LZMA-compressed header):** the next header is only a *descriptor*
|
||||||
|
whose `PackInfo` points at a packed header stream stored **in the middle** of the file (not
|
||||||
|
at EOF). Parse the descriptor's `StreamsInfo → kPackInfo (0x06)` to read `PackPos` and the
|
||||||
|
`PackSize`s (7z variable-length "numbers"; sum them). Ranged-read the contiguous packed
|
||||||
|
region `[32 + PackPos, Σ PackSize)` and add it as a **third** sparse region. `7z l` then
|
||||||
|
decodes the header from that region.
|
||||||
|
- **anything else:** return `null` (→ fallback).
|
||||||
|
4. `listFromSparse` with the 2 or 3 regions, runner = `7z l`. `parse7zOutput` yields
|
||||||
|
names+sizes (`crc32: null`, as today). Return `null` on bad magic / read failure / CLI error.
|
||||||
|
|
||||||
|
**Why the third region is required (spike finding, 2026-07-27):** the original two-region
|
||||||
|
(start+end) reconstruction was proven insufficient in a live test — `7z l` rejected it with
|
||||||
|
"Cannot open the file as [7z] archive" because these archives use an *encoded* header whose
|
||||||
|
compressed bytes live in a packed stream in the file body (a sparse hole), not at EOF. The
|
||||||
|
`0x17` branch fetches exactly that packed region. `7z l` still never touches the file-data
|
||||||
|
packed streams (it only lists), so those gaps stay sparse. The `read7zNumber` reader (7z's
|
||||||
|
base-128-ish variable-length integer with a first-byte length mask) and the minimal
|
||||||
|
`kPackInfo` walk are the only new 7z binary parsing; `7z l` still does the actual file listing.
|
||||||
|
All 7z placeholders are single-part.
|
||||||
|
|
||||||
|
### RAR ranged listing
|
||||||
|
|
||||||
|
RAR has no index; walk the block chain, parsing only each block's **size fields** to step
|
||||||
|
forward and harvest header bytes.
|
||||||
|
|
||||||
|
1. Read first ~16 bytes; detect **RAR4** (`52 61 72 21 1A 07 00`) vs **RAR5** (`…07 01 00`) and
|
||||||
|
the signature length.
|
||||||
|
2. From just after the signature, loop:
|
||||||
|
- Ranged-read a header chunk (start 8 KB; if parsed `HeaderSize` exceeds it — long filenames
|
||||||
|
— re-read exactly).
|
||||||
|
- Minimal block-extent parse:
|
||||||
|
- RAR5: `CRC32(4)` + vint `HeaderSize` + vint `HeaderType` + vint `HeaderFlags`; if the
|
||||||
|
"extra area" flag (`0x0001`) → vint `ExtraAreaSize`; if the "data present" flag
|
||||||
|
(`0x0002`) → vint `DataSize`. Next block = `pos + 4 + len(HeaderSize vint) + HeaderSize
|
||||||
|
+ DataSize`.
|
||||||
|
- RAR4: `HEAD_CRC(2)` + `HEAD_TYPE(1)` + `HEAD_FLAGS(2)` + `HEAD_SIZE(2)`; if flag `0x8000`
|
||||||
|
→ `ADD_SIZE(4)`. Next block = `pos + HEAD_SIZE + ADD_SIZE`.
|
||||||
|
- Harvest `[blockOffset, blockOffset + HeaderSize)` into the regions list.
|
||||||
|
- Stop at the end-of-archive block or EOF.
|
||||||
|
3. `listFromSparse` (headers present, data sparse) → `unrar lt` → `parseUnrarTechnical`. RAR
|
||||||
|
headers carry CRC32, so RAR contributes CRCs (fingerprint disambiguation keeps working).
|
||||||
|
|
||||||
|
**Multipart RAR** (1,016): each volume starts with its own signature + headers. Walk **each
|
||||||
|
part from its own signature**, reconstruct one sparse temp file per part with correct names
|
||||||
|
(`name.part1.rar`, `.part2.rar`, …) co-located in a temp dir, and run `unrar lt` on part 1 —
|
||||||
|
`unrar` auto-discovers co-located siblings (per the existing reader's note). The global-offset →
|
||||||
|
`(part, offsetInPart)` mapping reuses the multipart size math the ZIP path already uses.
|
||||||
|
|
||||||
|
## Fallback & integration
|
||||||
|
|
||||||
|
- A ranged reader returning `null` = cheap read failed (bad magic, read error, walk gave up, or
|
||||||
|
CLI error on the sparse file) → **full-download fallback**: download the whole archive, run the
|
||||||
|
existing `readRarContents` / `read7zContents`, backfill.
|
||||||
|
- The fallback is gated by `config.maxZipSizeMB` (the same guard used at ingest). Over the cap →
|
||||||
|
no download; write a `SystemNotification` (`INTEGRITY_AUDIT`, WARNING) and leave the listing
|
||||||
|
empty for manual review. This ensures nothing pathological (e.g. the 116 GB RAR) is pulled.
|
||||||
|
- Downstream is unchanged: `compareFingerprints` already treats null/incomplete CRCs as
|
||||||
|
"incomplete" (name-size path), and `backfillProvenance` writes entries when the candidate's
|
||||||
|
`fileCount === 0`.
|
||||||
|
- Observability: reuse the `zipsBackfilled` counter; add structured logs with
|
||||||
|
`confidence: "ranged" | "full-download-fallback"` and a WARN on fallback so miss-rate is
|
||||||
|
visible.
|
||||||
|
|
||||||
|
## Risks & de-risking spike (do before the full build)
|
||||||
|
|
||||||
|
On 3–4 real placeholder archives per format:
|
||||||
|
|
||||||
|
1. Confirm `downloadFileRange` returns correct bytes at **arbitrary (non-tail) offsets** —
|
||||||
|
currently only tail-verified in production. Underpins everything; if it fails, stop and
|
||||||
|
rethink. (Note: `range-download.ts` flags absolute-offset behaviour as pending live
|
||||||
|
verification; tail reads are proven by the 43 ZIP backfills done 2026-07-26.)
|
||||||
|
2. Confirm `7z l` and `unrar lt` list correctly from a **sparse reconstructed file** — single
|
||||||
|
part first, then multipart RAR (the highest-risk case).
|
||||||
|
|
||||||
|
If multipart-RAR sparse reconstruction proves unreliable in the spike, multipart RAR uses the
|
||||||
|
full-download fallback (respecting the size cap → oversized ones flagged, not downloaded).
|
||||||
|
|
||||||
|
## Testing
|
||||||
|
|
||||||
|
- **Unit (vitest, alongside `central-directory.test.ts`):** 7z signature-header parse; RAR4 &
|
||||||
|
RAR5 block-extent walk against committed small fixtures; `sparse-list` writes the correct
|
||||||
|
regions. Pure logic, no TDLib.
|
||||||
|
- **Live post-deploy:** watch `zipsBackfilled` climb for RAR/7z via the ranged path; spot-check
|
||||||
|
a handful of backfilled packages' `package_files` against a real `unrar lt` / `7z l` on a full
|
||||||
|
download of the same file; confirm the fallback/flag path fires on a deliberately-broken case.
|
||||||
|
|
||||||
|
## Rollout
|
||||||
|
|
||||||
|
Local build + deploy (no GitHub push required), per the established recipe: build
|
||||||
|
`worker/Dockerfile` locally, recreate the `dragonsstash-worker` container from the local image
|
||||||
|
(no `pull`). No new DB migration. The scheduler re-runs hourly and will backfill RAR/7z
|
||||||
|
placeholders on subsequent cycles.
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
-- AlterTable: track the forum topic currently being processed on the live run
|
||||||
|
-- so the worker status panel can offer a "skip & disable topic" action.
|
||||||
|
-- Additive, nullable — no data change for existing rows.
|
||||||
|
ALTER TABLE "ingestion_runs"
|
||||||
|
ADD COLUMN "currentTopicId" BIGINT,
|
||||||
|
ADD COLUMN "currentAccountChannelMapId" TEXT;
|
||||||
@@ -0,0 +1,5 @@
|
|||||||
|
-- AlterTable: count of packages whose provenance was backfilled during a run
|
||||||
|
-- (opportunistic cross-channel provenance backfill). Additive, non-null with a
|
||||||
|
-- default of 0 — no data change for existing rows.
|
||||||
|
ALTER TABLE "ingestion_runs"
|
||||||
|
ADD COLUMN "zipsBackfilled" INTEGER NOT NULL DEFAULT 0;
|
||||||
@@ -569,6 +569,7 @@ model IngestionRun {
|
|||||||
zipsFound Int @default(0)
|
zipsFound Int @default(0)
|
||||||
zipsDuplicate Int @default(0)
|
zipsDuplicate Int @default(0)
|
||||||
zipsIngested Int @default(0)
|
zipsIngested Int @default(0)
|
||||||
|
zipsBackfilled Int @default(0)
|
||||||
errorMessage String?
|
errorMessage String?
|
||||||
|
|
||||||
// Live activity tracking — written by worker in real-time
|
// Live activity tracking — written by worker in real-time
|
||||||
@@ -582,6 +583,8 @@ model IngestionRun {
|
|||||||
totalBytes BigInt? // Total size of current download
|
totalBytes BigInt? // Total size of current download
|
||||||
downloadPercent Int? // 0-100
|
downloadPercent Int? // 0-100
|
||||||
lastActivityAt DateTime? // When activity was last updated
|
lastActivityAt DateTime? // When activity was last updated
|
||||||
|
currentTopicId BigInt? // Forum topic currently being processed (live "skip topic")
|
||||||
|
currentAccountChannelMapId String? // AccountChannelMap owning the topic being processed
|
||||||
|
|
||||||
account TelegramAccount @relation(fields: [accountId], references: [id])
|
account TelegramAccount @relation(fields: [accountId], references: [id])
|
||||||
packages Package[]
|
packages Package[]
|
||||||
|
|||||||
@@ -1,7 +1,23 @@
|
|||||||
|
import { redirect } from "next/navigation";
|
||||||
|
import { auth } from "@/lib/auth";
|
||||||
|
import { prisma } from "@/lib/prisma";
|
||||||
import { Sidebar } from "@/components/layout/sidebar";
|
import { Sidebar } from "@/components/layout/sidebar";
|
||||||
import { Header } from "@/components/layout/header";
|
import { Header } from "@/components/layout/header";
|
||||||
|
|
||||||
export default function AppLayout({ children }: { children: React.ReactNode }) {
|
export default async function AppLayout({ children }: { children: React.ReactNode }) {
|
||||||
|
// Guard against a stale JWT session whose user no longer exists in the
|
||||||
|
// database (e.g. after a DB reset). The signed cookie still passes edge
|
||||||
|
// middleware, but every downstream query keyed on session.user.id would fail.
|
||||||
|
// Send such sessions to /logout, which clears the cookie and returns to login.
|
||||||
|
const session = await auth();
|
||||||
|
if (session?.user?.id) {
|
||||||
|
const user = await prisma.user.findUnique({
|
||||||
|
where: { id: session.user.id },
|
||||||
|
select: { id: true },
|
||||||
|
});
|
||||||
|
if (!user) redirect("/logout");
|
||||||
|
}
|
||||||
|
|
||||||
return (
|
return (
|
||||||
<div className="flex h-screen overflow-hidden">
|
<div className="flex h-screen overflow-hidden">
|
||||||
<div className="hidden lg:block">
|
<div className="hidden lg:block">
|
||||||
|
|||||||
@@ -0,0 +1,82 @@
|
|||||||
|
"use client";
|
||||||
|
|
||||||
|
import { useState } from "react";
|
||||||
|
import { Check, ChevronsUpDown } from "lucide-react";
|
||||||
|
import { cn } from "@/lib/utils";
|
||||||
|
import { Button } from "@/components/ui/button";
|
||||||
|
import {
|
||||||
|
Command,
|
||||||
|
CommandEmpty,
|
||||||
|
CommandGroup,
|
||||||
|
CommandInput,
|
||||||
|
CommandItem,
|
||||||
|
CommandList,
|
||||||
|
} from "@/components/ui/command";
|
||||||
|
import { Popover, PopoverContent, PopoverTrigger } from "@/components/ui/popover";
|
||||||
|
|
||||||
|
interface CreatorFilterProps {
|
||||||
|
creators: string[];
|
||||||
|
value: string; // active creator, "" when none
|
||||||
|
onChange: (creator: string) => void; // "" clears the filter
|
||||||
|
}
|
||||||
|
|
||||||
|
export function CreatorFilter({ creators, value, onChange }: CreatorFilterProps) {
|
||||||
|
const [open, setOpen] = useState(false);
|
||||||
|
|
||||||
|
return (
|
||||||
|
<Popover open={open} onOpenChange={setOpen}>
|
||||||
|
<PopoverTrigger asChild>
|
||||||
|
<Button
|
||||||
|
variant="outline"
|
||||||
|
size="sm"
|
||||||
|
role="combobox"
|
||||||
|
aria-expanded={open}
|
||||||
|
className="h-9 w-[200px] justify-between"
|
||||||
|
>
|
||||||
|
<span className="truncate">{value || "All Creators"}</span>
|
||||||
|
<ChevronsUpDown className="ml-2 h-4 w-4 shrink-0 opacity-50" />
|
||||||
|
</Button>
|
||||||
|
</PopoverTrigger>
|
||||||
|
<PopoverContent className="w-[240px] p-0" align="start">
|
||||||
|
<Command>
|
||||||
|
<CommandInput placeholder="Search creators..." className="h-9" />
|
||||||
|
<CommandList>
|
||||||
|
<CommandEmpty>No creators found.</CommandEmpty>
|
||||||
|
<CommandGroup>
|
||||||
|
<CommandItem
|
||||||
|
value="__all__"
|
||||||
|
onSelect={() => {
|
||||||
|
onChange("");
|
||||||
|
setOpen(false);
|
||||||
|
}}
|
||||||
|
>
|
||||||
|
<Check
|
||||||
|
className={cn("mr-2 h-4 w-4", value === "" ? "opacity-100" : "opacity-0")}
|
||||||
|
/>
|
||||||
|
All Creators
|
||||||
|
</CommandItem>
|
||||||
|
{creators.map((creator) => (
|
||||||
|
<CommandItem
|
||||||
|
key={creator}
|
||||||
|
value={creator}
|
||||||
|
onSelect={() => {
|
||||||
|
onChange(creator);
|
||||||
|
setOpen(false);
|
||||||
|
}}
|
||||||
|
>
|
||||||
|
<Check
|
||||||
|
className={cn(
|
||||||
|
"mr-2 h-4 w-4",
|
||||||
|
value === creator ? "opacity-100" : "opacity-0"
|
||||||
|
)}
|
||||||
|
/>
|
||||||
|
<span className="truncate">{creator}</span>
|
||||||
|
</CommandItem>
|
||||||
|
))}
|
||||||
|
</CommandGroup>
|
||||||
|
</CommandList>
|
||||||
|
</Command>
|
||||||
|
</PopoverContent>
|
||||||
|
</Popover>
|
||||||
|
);
|
||||||
|
}
|
||||||
@@ -3,7 +3,7 @@
|
|||||||
import { useState, useCallback, useTransition, useMemo, useRef } from "react";
|
import { useState, useCallback, useTransition, useMemo, useRef } from "react";
|
||||||
import { useRouter, usePathname, useSearchParams } from "next/navigation";
|
import { useRouter, usePathname, useSearchParams } from "next/navigation";
|
||||||
import { toast } from "sonner";
|
import { toast } from "sonner";
|
||||||
import { Search, Layers, Upload } from "lucide-react";
|
import { Search, Layers, Upload, Send } from "lucide-react";
|
||||||
import { UploadDialog } from "./upload-dialog";
|
import { UploadDialog } from "./upload-dialog";
|
||||||
import { useDataTable } from "@/hooks/use-data-table";
|
import { useDataTable } from "@/hooks/use-data-table";
|
||||||
import {
|
import {
|
||||||
@@ -16,6 +16,7 @@ import {
|
|||||||
import { PackageFilesDrawer } from "./package-files-drawer";
|
import { PackageFilesDrawer } from "./package-files-drawer";
|
||||||
import { IngestionStatus } from "./ingestion-status";
|
import { IngestionStatus } from "./ingestion-status";
|
||||||
import { SkippedPackagesTab } from "./skipped-packages-tab";
|
import { SkippedPackagesTab } from "./skipped-packages-tab";
|
||||||
|
import { CreatorFilter } from "./creator-filter";
|
||||||
import { DataTable } from "@/components/shared/data-table";
|
import { DataTable } from "@/components/shared/data-table";
|
||||||
import { DataTablePagination } from "@/components/shared/data-table-pagination";
|
import { DataTablePagination } from "@/components/shared/data-table-pagination";
|
||||||
import { DataTableViewOptions } from "@/components/shared/data-table-view-options";
|
import { DataTableViewOptions } from "@/components/shared/data-table-view-options";
|
||||||
@@ -49,6 +50,7 @@ import {
|
|||||||
createGroupAction,
|
createGroupAction,
|
||||||
removeFromGroupAction,
|
removeFromGroupAction,
|
||||||
sendAllInGroupAction,
|
sendAllInGroupAction,
|
||||||
|
sendAllFromCreatorAction,
|
||||||
updateGroupPreviewAction,
|
updateGroupPreviewAction,
|
||||||
mergeGroupsAction,
|
mergeGroupsAction,
|
||||||
} from "../actions";
|
} from "../actions";
|
||||||
@@ -59,6 +61,7 @@ interface StlTableProps {
|
|||||||
totalCount: number;
|
totalCount: number;
|
||||||
ingestionStatus: IngestionAccountStatus[];
|
ingestionStatus: IngestionAccountStatus[];
|
||||||
availableTags: string[];
|
availableTags: string[];
|
||||||
|
availableCreators: string[];
|
||||||
searchTerm: string;
|
searchTerm: string;
|
||||||
skippedData: SkippedRow[];
|
skippedData: SkippedRow[];
|
||||||
skippedPageCount: number;
|
skippedPageCount: number;
|
||||||
@@ -74,6 +77,7 @@ export function StlTable({
|
|||||||
totalCount,
|
totalCount,
|
||||||
ingestionStatus,
|
ingestionStatus,
|
||||||
availableTags,
|
availableTags,
|
||||||
|
availableCreators,
|
||||||
searchTerm,
|
searchTerm,
|
||||||
skippedData,
|
skippedData,
|
||||||
skippedPageCount,
|
skippedPageCount,
|
||||||
@@ -85,6 +89,7 @@ export function StlTable({
|
|||||||
const router = useRouter();
|
const router = useRouter();
|
||||||
const pathname = usePathname();
|
const pathname = usePathname();
|
||||||
const searchParams = useSearchParams();
|
const searchParams = useSearchParams();
|
||||||
|
const activeCreator = searchParams.get("creator") ?? "";
|
||||||
|
|
||||||
const [searchValue, setSearchValue] = useState(searchParams.get("search") ?? "");
|
const [searchValue, setSearchValue] = useState(searchParams.get("search") ?? "");
|
||||||
const [viewPkg, setViewPkg] = useState<PackageRow | null>(null);
|
const [viewPkg, setViewPkg] = useState<PackageRow | null>(null);
|
||||||
@@ -207,6 +212,20 @@ export function StlTable({
|
|||||||
[router, pathname, searchParams]
|
[router, pathname, searchParams]
|
||||||
);
|
);
|
||||||
|
|
||||||
|
const updateCreatorFilter = useCallback(
|
||||||
|
(value: string) => {
|
||||||
|
const params = new URLSearchParams(searchParams.toString());
|
||||||
|
if (value) {
|
||||||
|
params.set("creator", value);
|
||||||
|
params.set("page", "1");
|
||||||
|
} else {
|
||||||
|
params.delete("creator");
|
||||||
|
}
|
||||||
|
router.push(`${pathname}?${params.toString()}`, { scroll: false });
|
||||||
|
},
|
||||||
|
[router, pathname, searchParams]
|
||||||
|
);
|
||||||
|
|
||||||
const activeTab = searchParams.get("tab") ?? "packages";
|
const activeTab = searchParams.get("tab") ?? "packages";
|
||||||
|
|
||||||
const updateTab = useCallback(
|
const updateTab = useCallback(
|
||||||
@@ -277,6 +296,23 @@ export function StlTable({
|
|||||||
[router]
|
[router]
|
||||||
);
|
);
|
||||||
|
|
||||||
|
const handleSendAllFromCreator = useCallback(() => {
|
||||||
|
if (!confirm(`Send all packages from "${activeCreator}" to your Telegram?`)) return;
|
||||||
|
startTransition(async () => {
|
||||||
|
const result = await sendAllFromCreatorAction(activeCreator);
|
||||||
|
if (result.success) {
|
||||||
|
const { queued, skipped } = result.data;
|
||||||
|
toast.success(
|
||||||
|
`Queued ${queued} package${queued === 1 ? "" : "s"} from ${activeCreator}` +
|
||||||
|
(skipped ? ` (${skipped} already queued)` : "")
|
||||||
|
);
|
||||||
|
router.refresh();
|
||||||
|
} else {
|
||||||
|
toast.error(result.error);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}, [activeCreator, router]);
|
||||||
|
|
||||||
const handleRemoveFromGroup = useCallback(
|
const handleRemoveFromGroup = useCallback(
|
||||||
(packageId: string) => {
|
(packageId: string) => {
|
||||||
startTransition(async () => {
|
startTransition(async () => {
|
||||||
@@ -464,14 +500,9 @@ export function StlTable({
|
|||||||
</Badge>
|
</Badge>
|
||||||
)}
|
)}
|
||||||
</TabsTrigger>
|
</TabsTrigger>
|
||||||
<TabsTrigger value="ungrouped" className="gap-1.5">
|
{/* "Ungrouped" tab hidden: the STL list is now flat and grouping is
|
||||||
Ungrouped
|
no longer surfaced here. The tab content below is kept (unreachable)
|
||||||
{ungroupedTotalCount > 0 && (
|
to avoid churn; remove it and its data fetch if grouping is dropped. */}
|
||||||
<Badge variant="secondary" className="h-5 px-1.5 text-[10px]">
|
|
||||||
{ungroupedTotalCount}
|
|
||||||
</Badge>
|
|
||||||
)}
|
|
||||||
</TabsTrigger>
|
|
||||||
</TabsList>
|
</TabsList>
|
||||||
|
|
||||||
<TabsContent value="packages" className="space-y-4">
|
<TabsContent value="packages" className="space-y-4">
|
||||||
@@ -500,11 +531,29 @@ export function StlTable({
|
|||||||
</SelectContent>
|
</SelectContent>
|
||||||
</Select>
|
</Select>
|
||||||
)}
|
)}
|
||||||
|
{availableCreators.length > 0 && (
|
||||||
|
<CreatorFilter
|
||||||
|
creators={availableCreators}
|
||||||
|
value={activeCreator}
|
||||||
|
onChange={updateCreatorFilter}
|
||||||
|
/>
|
||||||
|
)}
|
||||||
<DataTableViewOptions table={table} />
|
<DataTableViewOptions table={table} />
|
||||||
<Button variant="outline" size="sm" className="h-9" onClick={() => setUploadOpen(true)}>
|
<Button variant="outline" size="sm" className="h-9" onClick={() => setUploadOpen(true)}>
|
||||||
<Upload className="mr-2 h-4 w-4" />
|
<Upload className="mr-2 h-4 w-4" />
|
||||||
Upload Files
|
Upload Files
|
||||||
</Button>
|
</Button>
|
||||||
|
{activeCreator && (
|
||||||
|
<Button
|
||||||
|
variant="outline"
|
||||||
|
size="sm"
|
||||||
|
className="h-9 gap-1.5"
|
||||||
|
onClick={handleSendAllFromCreator}
|
||||||
|
>
|
||||||
|
<Send className="h-3.5 w-3.5" />
|
||||||
|
Send all from {activeCreator}
|
||||||
|
</Button>
|
||||||
|
)}
|
||||||
{selectedPackages.size >= 2 && (
|
{selectedPackages.size >= 2 && (
|
||||||
<Button
|
<Button
|
||||||
variant="outline"
|
variant="outline"
|
||||||
|
|||||||
@@ -589,3 +589,82 @@ export async function sendAllInGroupAction(
|
|||||||
return { success: false, error: "Failed to send group packages" };
|
return { success: false, error: "Failed to send group packages" };
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
export async function sendAllFromCreatorAction(
|
||||||
|
creatorName: string
|
||||||
|
): Promise<ActionResult<{ queued: number; skipped: number }>> {
|
||||||
|
const session = await auth();
|
||||||
|
if (!session?.user?.id) return { success: false, error: "Unauthorized" };
|
||||||
|
|
||||||
|
const creator = creatorName.trim();
|
||||||
|
if (!creator) {
|
||||||
|
return { success: false, error: "No creator specified" };
|
||||||
|
}
|
||||||
|
|
||||||
|
try {
|
||||||
|
const telegramLink = await prisma.telegramLink.findUnique({
|
||||||
|
where: { userId: session.user.id },
|
||||||
|
});
|
||||||
|
|
||||||
|
if (!telegramLink) {
|
||||||
|
return { success: false, error: "No linked Telegram account. Link one in Settings." };
|
||||||
|
}
|
||||||
|
|
||||||
|
const sendablePackages = await prisma.package.findMany({
|
||||||
|
where: {
|
||||||
|
creator,
|
||||||
|
destChannelId: { not: null },
|
||||||
|
destMessageId: { not: null },
|
||||||
|
},
|
||||||
|
select: { id: true },
|
||||||
|
});
|
||||||
|
|
||||||
|
if (sendablePackages.length === 0) {
|
||||||
|
return { success: false, error: "No uploaded packages found for this creator" };
|
||||||
|
}
|
||||||
|
|
||||||
|
let queued = 0;
|
||||||
|
let skipped = 0;
|
||||||
|
for (const pkg of sendablePackages) {
|
||||||
|
// Only create if no existing PENDING/SENDING request for this package+link combo
|
||||||
|
const existing = await prisma.botSendRequest.findFirst({
|
||||||
|
where: {
|
||||||
|
packageId: pkg.id,
|
||||||
|
telegramLinkId: telegramLink.id,
|
||||||
|
status: { in: ["PENDING", "SENDING"] },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
if (existing) {
|
||||||
|
skipped++;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
const sendRequest = await prisma.botSendRequest.create({
|
||||||
|
data: {
|
||||||
|
packageId: pkg.id,
|
||||||
|
telegramLinkId: telegramLink.id,
|
||||||
|
requestedByUserId: session.user.id,
|
||||||
|
status: "PENDING",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
// Notify the bot via pg_notify
|
||||||
|
try {
|
||||||
|
await prisma.$queryRawUnsafe(
|
||||||
|
`SELECT pg_notify('bot_send', $1)`,
|
||||||
|
sendRequest.id
|
||||||
|
);
|
||||||
|
} catch {
|
||||||
|
// Best-effort — the bot also polls periodically
|
||||||
|
}
|
||||||
|
|
||||||
|
queued++;
|
||||||
|
}
|
||||||
|
|
||||||
|
revalidatePath("/stls");
|
||||||
|
return { success: true, data: { queued, skipped } };
|
||||||
|
} catch {
|
||||||
|
return { success: false, error: "Failed to send creator packages" };
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
import { auth } from "@/lib/auth";
|
import { auth } from "@/lib/auth";
|
||||||
import { redirect } from "next/navigation";
|
import { redirect } from "next/navigation";
|
||||||
import { listDisplayItems, searchPackages, getIngestionStatus, getAllPackageTags, listSkippedPackages, countSkippedPackages, listUngroupedPackages, countUngroupedPackages } from "@/lib/telegram/queries";
|
import { listDisplayItems, searchPackages, getIngestionStatus, getAllPackageTags, getAllPackageCreators, listSkippedPackages, countSkippedPackages, listUngroupedPackages, countUngroupedPackages } from "@/lib/telegram/queries";
|
||||||
import { StlTable } from "./_components/stl-table";
|
import { StlTable } from "./_components/stl-table";
|
||||||
import type { DisplayItem, PackageListItem } from "@/lib/telegram/types";
|
import type { DisplayItem, PackageListItem } from "@/lib/telegram/types";
|
||||||
|
|
||||||
@@ -24,7 +24,7 @@ export default async function StlFilesPage({ searchParams }: Props) {
|
|||||||
const tab = (params.tab as string) ?? "packages";
|
const tab = (params.tab as string) ?? "packages";
|
||||||
|
|
||||||
// Fetch packages, ingestion status, tags, and skipped count in parallel
|
// Fetch packages, ingestion status, tags, and skipped count in parallel
|
||||||
const [result, ingestionStatus, availableTags, skippedCount, ungroupedCount] = await Promise.all([
|
const [result, ingestionStatus, availableTags, availableCreators, skippedCount, ungroupedCount] = await Promise.all([
|
||||||
search
|
search
|
||||||
? searchPackages({
|
? searchPackages({
|
||||||
query: search,
|
query: search,
|
||||||
@@ -42,6 +42,7 @@ export default async function StlFilesPage({ searchParams }: Props) {
|
|||||||
}),
|
}),
|
||||||
getIngestionStatus(),
|
getIngestionStatus(),
|
||||||
getAllPackageTags(),
|
getAllPackageTags(),
|
||||||
|
getAllPackageCreators(),
|
||||||
countSkippedPackages(),
|
countSkippedPackages(),
|
||||||
countUngroupedPackages(),
|
countUngroupedPackages(),
|
||||||
]);
|
]);
|
||||||
@@ -68,6 +69,7 @@ export default async function StlFilesPage({ searchParams }: Props) {
|
|||||||
totalCount={result.pagination.total}
|
totalCount={result.pagination.total}
|
||||||
ingestionStatus={ingestionStatus}
|
ingestionStatus={ingestionStatus}
|
||||||
availableTags={availableTags}
|
availableTags={availableTags}
|
||||||
|
availableCreators={availableCreators}
|
||||||
searchTerm={search}
|
searchTerm={search}
|
||||||
skippedData={skippedResult?.items ?? []}
|
skippedData={skippedResult?.items ?? []}
|
||||||
skippedPageCount={skippedResult?.pagination.totalPages ?? 0}
|
skippedPageCount={skippedResult?.pagination.totalPages ?? 0}
|
||||||
|
|||||||
@@ -9,13 +9,14 @@ import {
|
|||||||
Radio,
|
Radio,
|
||||||
AlertTriangle,
|
AlertTriangle,
|
||||||
RefreshCw,
|
RefreshCw,
|
||||||
|
SkipForward,
|
||||||
} from "lucide-react";
|
} from "lucide-react";
|
||||||
import { Card, CardContent } from "@/components/ui/card";
|
import { Card, CardContent } from "@/components/ui/card";
|
||||||
import { Badge } from "@/components/ui/badge";
|
import { Badge } from "@/components/ui/badge";
|
||||||
import { Button } from "@/components/ui/button";
|
import { Button } from "@/components/ui/button";
|
||||||
import { cn } from "@/lib/utils";
|
import { cn } from "@/lib/utils";
|
||||||
import { toast } from "sonner";
|
import { toast } from "sonner";
|
||||||
import { triggerIngestion } from "../actions";
|
import { triggerIngestion, disableActiveTopic } from "../actions";
|
||||||
import type { IngestionAccountStatus } from "@/lib/telegram/types";
|
import type { IngestionAccountStatus } from "@/lib/telegram/types";
|
||||||
|
|
||||||
interface WorkerStatusPanelProps {
|
interface WorkerStatusPanelProps {
|
||||||
@@ -218,6 +219,24 @@ function RunningStatus({
|
|||||||
}: {
|
}: {
|
||||||
run: NonNullable<IngestionAccountStatus["currentRun"]>;
|
run: NonNullable<IngestionAccountStatus["currentRun"]>;
|
||||||
}) {
|
}) {
|
||||||
|
const [isDisabling, startDisable] = useTransition();
|
||||||
|
|
||||||
|
const handleSkipTopic = () => {
|
||||||
|
const acmId = run.currentAccountChannelMapId;
|
||||||
|
const topicId = run.currentTopicId;
|
||||||
|
if (!acmId || !topicId) return;
|
||||||
|
startDisable(async () => {
|
||||||
|
const result = await disableActiveTopic(acmId, topicId);
|
||||||
|
if (result.success) {
|
||||||
|
toast.success(
|
||||||
|
"Topic disabled — the current file finishes, then the rest is skipped"
|
||||||
|
);
|
||||||
|
} else {
|
||||||
|
toast.error(result.error ?? "Failed to disable topic");
|
||||||
|
}
|
||||||
|
});
|
||||||
|
};
|
||||||
|
|
||||||
return (
|
return (
|
||||||
<div className="space-y-2">
|
<div className="space-y-2">
|
||||||
<div className="flex items-center gap-2">
|
<div className="flex items-center gap-2">
|
||||||
@@ -273,6 +292,27 @@ function RunningStatus({
|
|||||||
</span>
|
</span>
|
||||||
)}
|
)}
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
|
{/* Skip & disable the topic currently being processed */}
|
||||||
|
{run.currentTopicId && run.currentAccountChannelMapId && (
|
||||||
|
<div className="pl-6 pt-1">
|
||||||
|
<Button
|
||||||
|
variant="outline"
|
||||||
|
size="sm"
|
||||||
|
className="h-7 gap-1.5 text-xs"
|
||||||
|
onClick={handleSkipTopic}
|
||||||
|
disabled={isDisabling}
|
||||||
|
title="Finish the current file, skip the rest of this topic, and don't fetch it in future runs"
|
||||||
|
>
|
||||||
|
{isDisabling ? (
|
||||||
|
<Loader2 className="h-3 w-3 animate-spin" />
|
||||||
|
) : (
|
||||||
|
<SkipForward className="h-3 w-3" />
|
||||||
|
)}
|
||||||
|
Skip & disable this topic
|
||||||
|
</Button>
|
||||||
|
</div>
|
||||||
|
)}
|
||||||
</div>
|
</div>
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -462,6 +462,54 @@ export async function setTopicFetchEnabled(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Disable the topic currently being processed by the worker, identified by its
|
||||||
|
* account-channel mapping + Telegram topic id (as exposed on the live run
|
||||||
|
* status). Upserts so a disabled row exists even if the worker hasn't persisted
|
||||||
|
* the topic yet. The worker honours this live: it finishes the in-flight file
|
||||||
|
* then skips the rest of the topic, and future runs skip it entirely.
|
||||||
|
*/
|
||||||
|
export async function disableActiveTopic(
|
||||||
|
accountChannelMapId: string,
|
||||||
|
topicId: string
|
||||||
|
): Promise<ActionResult> {
|
||||||
|
const admin = await requireAdmin();
|
||||||
|
if (!admin.success) return admin;
|
||||||
|
|
||||||
|
let topicIdBig: bigint;
|
||||||
|
try {
|
||||||
|
topicIdBig = BigInt(topicId);
|
||||||
|
} catch {
|
||||||
|
return { success: false, error: "Invalid topic id" };
|
||||||
|
}
|
||||||
|
// BigInt("") / BigInt("0") don't throw — reject non-positive ids explicitly
|
||||||
|
// since this action is exported and admin-callable outside the UI button.
|
||||||
|
if (topicIdBig <= BigInt(0)) {
|
||||||
|
return { success: false, error: "Invalid topic id" };
|
||||||
|
}
|
||||||
|
|
||||||
|
try {
|
||||||
|
await prisma.topicProgress.upsert({
|
||||||
|
where: {
|
||||||
|
accountChannelMapId_topicId: {
|
||||||
|
accountChannelMapId,
|
||||||
|
topicId: topicIdBig,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
create: {
|
||||||
|
accountChannelMapId,
|
||||||
|
topicId: topicIdBig,
|
||||||
|
fetchEnabled: false,
|
||||||
|
},
|
||||||
|
update: { fetchEnabled: false },
|
||||||
|
});
|
||||||
|
revalidatePath(REVALIDATE_PATH);
|
||||||
|
return { success: true, data: undefined };
|
||||||
|
} catch {
|
||||||
|
return { success: false, error: "Failed to disable topic" };
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ── Account-Channel link actions ──
|
// ── Account-Channel link actions ──
|
||||||
|
|
||||||
export async function linkChannel(
|
export async function linkChannel(
|
||||||
|
|||||||
@@ -0,0 +1,9 @@
|
|||||||
|
import { signOut } from "@/lib/auth";
|
||||||
|
|
||||||
|
// Server-side sign-out that clears the JWT session cookie and redirects to the
|
||||||
|
// login page. Used to recover from a stale session whose user no longer exists
|
||||||
|
// in the database (e.g. after a DB reset), which a client-only signOut can't
|
||||||
|
// reach because the app crashes before rendering the user menu.
|
||||||
|
export async function GET() {
|
||||||
|
await signOut({ redirectTo: "/login" });
|
||||||
|
}
|
||||||
@@ -1,20 +1,37 @@
|
|||||||
|
import { Prisma } from "@prisma/client";
|
||||||
import { prisma } from "@/lib/prisma";
|
import { prisma } from "@/lib/prisma";
|
||||||
|
|
||||||
|
const DEFAULT_SETTINGS = {
|
||||||
|
lowStockThreshold: 20,
|
||||||
|
currency: "EUR",
|
||||||
|
theme: "dark",
|
||||||
|
units: "metric",
|
||||||
|
} as const;
|
||||||
|
|
||||||
export async function getUserSettings(userId: string) {
|
export async function getUserSettings(userId: string) {
|
||||||
let settings = await prisma.userSettings.findUnique({
|
let settings = await prisma.userSettings.findUnique({
|
||||||
where: { userId },
|
where: { userId },
|
||||||
});
|
});
|
||||||
|
|
||||||
if (!settings) {
|
if (!settings) {
|
||||||
settings = await prisma.userSettings.create({
|
try {
|
||||||
data: {
|
settings = await prisma.userSettings.create({
|
||||||
userId,
|
data: { userId, ...DEFAULT_SETTINGS },
|
||||||
lowStockThreshold: 20,
|
});
|
||||||
currency: "EUR",
|
} catch (err) {
|
||||||
theme: "dark",
|
// The session's user may no longer exist (e.g. a stale JWT cookie after a
|
||||||
units: "metric",
|
// database reset). Creating settings then hits a foreign-key violation
|
||||||
},
|
// (P2003). Don't crash the Server Component render — return unsaved
|
||||||
});
|
// defaults. The (app) layout guard redirects such stale sessions to
|
||||||
|
// sign-out, so this fallback is only ever momentarily visible.
|
||||||
|
if (
|
||||||
|
err instanceof Prisma.PrismaClientKnownRequestError &&
|
||||||
|
err.code === "P2003"
|
||||||
|
) {
|
||||||
|
return { id: "", userId, ...DEFAULT_SETTINGS };
|
||||||
|
}
|
||||||
|
throw err;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
return settings;
|
return settings;
|
||||||
|
|||||||
+79
-24
@@ -9,6 +9,28 @@ import type {
|
|||||||
PackageGroupRow,
|
PackageGroupRow,
|
||||||
} from "./types";
|
} from "./types";
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Returns the subset of the given IDs whose `previewData` column is non-null,
|
||||||
|
* WITHOUT transferring the (large) image bytes.
|
||||||
|
*
|
||||||
|
* List views only need a `hasPreview` boolean per row. Selecting `previewData`
|
||||||
|
* directly pulls every JPEG blob (avg ~700 KB, up to 2 MB) into the Node heap
|
||||||
|
* just to compare it against null — under concurrent requests this exhausts the
|
||||||
|
* container memory limit and crashes the process. This keeps the check in SQL.
|
||||||
|
*/
|
||||||
|
async function fetchPreviewFlags(
|
||||||
|
table: "packages" | "package_groups",
|
||||||
|
ids: string[]
|
||||||
|
): Promise<Set<string>> {
|
||||||
|
if (ids.length === 0) return new Set();
|
||||||
|
const placeholders = ids.map((_, i) => `$${i + 1}`).join(", ");
|
||||||
|
const rows = await prisma.$queryRawUnsafe<{ id: string }[]>(
|
||||||
|
`SELECT id FROM ${table} WHERE "previewData" IS NOT NULL AND id IN (${placeholders})`,
|
||||||
|
...ids
|
||||||
|
);
|
||||||
|
return new Set(rows.map((r) => r.id));
|
||||||
|
}
|
||||||
|
|
||||||
export async function listPackages(options: {
|
export async function listPackages(options: {
|
||||||
page: number;
|
page: number;
|
||||||
limit: number;
|
limit: number;
|
||||||
@@ -40,13 +62,17 @@ export async function listPackages(options: {
|
|||||||
indexedAt: true,
|
indexedAt: true,
|
||||||
creator: true,
|
creator: true,
|
||||||
tags: true,
|
tags: true,
|
||||||
previewData: true, // check actual image data, not previewMsgId proxy
|
|
||||||
sourceChannel: { select: { id: true, title: true } },
|
sourceChannel: { select: { id: true, title: true } },
|
||||||
},
|
},
|
||||||
}),
|
}),
|
||||||
prisma.package.count({ where }),
|
prisma.package.count({ where }),
|
||||||
]);
|
]);
|
||||||
|
|
||||||
|
const previewIds = await fetchPreviewFlags(
|
||||||
|
"packages",
|
||||||
|
items.map((p) => p.id)
|
||||||
|
);
|
||||||
|
|
||||||
const mapped: PackageListItem[] = items.map((pkg) => ({
|
const mapped: PackageListItem[] = items.map((pkg) => ({
|
||||||
id: pkg.id,
|
id: pkg.id,
|
||||||
fileName: pkg.fileName,
|
fileName: pkg.fileName,
|
||||||
@@ -55,7 +81,7 @@ export async function listPackages(options: {
|
|||||||
archiveType: pkg.archiveType,
|
archiveType: pkg.archiveType,
|
||||||
fileCount: pkg.fileCount,
|
fileCount: pkg.fileCount,
|
||||||
isMultipart: pkg.isMultipart,
|
isMultipart: pkg.isMultipart,
|
||||||
hasPreview: pkg.previewData !== null,
|
hasPreview: previewIds.has(pkg.id),
|
||||||
creator: pkg.creator,
|
creator: pkg.creator,
|
||||||
tags: pkg.tags,
|
tags: pkg.tags,
|
||||||
indexedAt: pkg.indexedAt.toISOString(),
|
indexedAt: pkg.indexedAt.toISOString(),
|
||||||
@@ -109,31 +135,31 @@ export async function listDisplayItems(options: {
|
|||||||
const sortCol = sortBy === "fileName" ? `"fileName"` : sortBy === "fileSize" ? `"fileSize"` : `"indexedAt"`;
|
const sortCol = sortBy === "fileName" ? `"fileName"` : sortBy === "fileSize" ? `"fileSize"` : `"indexedAt"`;
|
||||||
const sortDir = order === "asc" ? "ASC" : "DESC";
|
const sortDir = order === "asc" ? "ASC" : "DESC";
|
||||||
|
|
||||||
// Step 1: Count display items
|
// NOTE: The STL list is intentionally FLAT — every package is its own display
|
||||||
|
// row regardless of packageGroupId. Grouping is no longer surfaced in this
|
||||||
|
// view (the creator column/filter organizes the list instead). PackageGroup
|
||||||
|
// rows and the manual grouping actions still exist in the DB/UI; they just
|
||||||
|
// don't drive this list's layout anymore.
|
||||||
|
|
||||||
|
// Step 1: Count display items (one per package)
|
||||||
const countResult = await prisma.$queryRawUnsafe<[{ count: bigint }]>(
|
const countResult = await prisma.$queryRawUnsafe<[{ count: bigint }]>(
|
||||||
`SELECT COUNT(*) AS count FROM (
|
`SELECT COUNT(*) AS count FROM packages p ${whereClause}`,
|
||||||
SELECT DISTINCT COALESCE(p."packageGroupId", p."id") AS display_id
|
|
||||||
FROM packages p
|
|
||||||
${whereClause}
|
|
||||||
) AS display_items`,
|
|
||||||
...params
|
...params
|
||||||
);
|
);
|
||||||
const total = Number(countResult[0].count);
|
const total = Number(countResult[0].count);
|
||||||
|
|
||||||
// Step 2: Get display item IDs for this page
|
// Step 2: Get package IDs for this page
|
||||||
const limitParam = paramIdx++;
|
const limitParam = paramIdx++;
|
||||||
const offsetParam = paramIdx++;
|
const offsetParam = paramIdx++;
|
||||||
const displayRows = await prisma.$queryRawUnsafe<
|
const displayRows = await prisma.$queryRawUnsafe<
|
||||||
{ display_id: string; display_type: string }[]
|
{ display_id: string; display_type: string }[]
|
||||||
>(
|
>(
|
||||||
`SELECT
|
`SELECT
|
||||||
COALESCE(p."packageGroupId", p."id") AS display_id,
|
p."id" AS display_id,
|
||||||
CASE WHEN p."packageGroupId" IS NOT NULL THEN 'group' ELSE 'package' END AS display_type,
|
'package' AS display_type,
|
||||||
MAX(p.${sortCol}) AS sort_value
|
p.${sortCol} AS sort_value
|
||||||
FROM packages p
|
FROM packages p
|
||||||
${whereClause}
|
${whereClause}
|
||||||
GROUP BY COALESCE(p."packageGroupId", p."id"),
|
|
||||||
CASE WHEN p."packageGroupId" IS NOT NULL THEN 'group' ELSE 'package' END
|
|
||||||
ORDER BY sort_value ${sortDir}
|
ORDER BY sort_value ${sortDir}
|
||||||
LIMIT $${limitParam} OFFSET $${offsetParam}`,
|
LIMIT $${limitParam} OFFSET $${offsetParam}`,
|
||||||
...params, limit, (page - 1) * limit
|
...params, limit, (page - 1) * limit
|
||||||
@@ -149,7 +175,7 @@ export async function listDisplayItems(options: {
|
|||||||
select: {
|
select: {
|
||||||
id: true, fileName: true, fileSize: true, contentHash: true,
|
id: true, fileName: true, fileSize: true, contentHash: true,
|
||||||
archiveType: true, fileCount: true, isMultipart: true,
|
archiveType: true, fileCount: true, isMultipart: true,
|
||||||
indexedAt: true, creator: true, tags: true, previewData: true,
|
indexedAt: true, creator: true, tags: true,
|
||||||
sourceChannel: { select: { id: true, title: true } },
|
sourceChannel: { select: { id: true, title: true } },
|
||||||
},
|
},
|
||||||
})
|
})
|
||||||
@@ -159,13 +185,13 @@ export async function listDisplayItems(options: {
|
|||||||
? await prisma.packageGroup.findMany({
|
? await prisma.packageGroup.findMany({
|
||||||
where: { id: { in: groupIds } },
|
where: { id: { in: groupIds } },
|
||||||
select: {
|
select: {
|
||||||
id: true, name: true, previewData: true,
|
id: true, name: true,
|
||||||
sourceChannel: { select: { id: true, title: true } },
|
sourceChannel: { select: { id: true, title: true } },
|
||||||
packages: {
|
packages: {
|
||||||
select: {
|
select: {
|
||||||
id: true, fileName: true, fileSize: true, contentHash: true,
|
id: true, fileName: true, fileSize: true, contentHash: true,
|
||||||
archiveType: true, fileCount: true, isMultipart: true,
|
archiveType: true, fileCount: true, isMultipart: true,
|
||||||
indexedAt: true, creator: true, tags: true, previewData: true,
|
indexedAt: true, creator: true, tags: true,
|
||||||
sourceChannel: { select: { id: true, title: true } },
|
sourceChannel: { select: { id: true, title: true } },
|
||||||
},
|
},
|
||||||
orderBy: { indexedAt: "desc" },
|
orderBy: { indexedAt: "desc" },
|
||||||
@@ -174,6 +200,16 @@ export async function listDisplayItems(options: {
|
|||||||
})
|
})
|
||||||
: [];
|
: [];
|
||||||
|
|
||||||
|
// Compute hasPreview flags without transferring image bytes (see fetchPreviewFlags)
|
||||||
|
const allPackageIds = [
|
||||||
|
...standalonePackages.map((p) => p.id),
|
||||||
|
...groups.flatMap((g) => g.packages.map((p) => p.id)),
|
||||||
|
];
|
||||||
|
const [packagePreviewIds, groupPreviewIds] = await Promise.all([
|
||||||
|
fetchPreviewFlags("packages", allPackageIds),
|
||||||
|
fetchPreviewFlags("package_groups", groups.map((g) => g.id)),
|
||||||
|
]);
|
||||||
|
|
||||||
// Build DisplayItem array in the original sort order
|
// Build DisplayItem array in the original sort order
|
||||||
const packageMap = new Map(standalonePackages.map((p) => [p.id, p]));
|
const packageMap = new Map(standalonePackages.map((p) => [p.id, p]));
|
||||||
const groupMap = new Map(groups.map((g) => [g.id, g]));
|
const groupMap = new Map(groups.map((g) => [g.id, g]));
|
||||||
@@ -191,7 +227,7 @@ export async function listDisplayItems(options: {
|
|||||||
archiveType: pkg.archiveType,
|
archiveType: pkg.archiveType,
|
||||||
fileCount: pkg.fileCount,
|
fileCount: pkg.fileCount,
|
||||||
isMultipart: pkg.isMultipart,
|
isMultipart: pkg.isMultipart,
|
||||||
hasPreview: pkg.previewData !== null,
|
hasPreview: packagePreviewIds.has(pkg.id),
|
||||||
creator: pkg.creator,
|
creator: pkg.creator,
|
||||||
tags: pkg.tags,
|
tags: pkg.tags,
|
||||||
indexedAt: pkg.indexedAt.toISOString(),
|
indexedAt: pkg.indexedAt.toISOString(),
|
||||||
@@ -209,7 +245,7 @@ export async function listDisplayItems(options: {
|
|||||||
data: {
|
data: {
|
||||||
id: grp.id,
|
id: grp.id,
|
||||||
name: grp.name,
|
name: grp.name,
|
||||||
hasPreview: grp.previewData !== null,
|
hasPreview: groupPreviewIds.has(grp.id),
|
||||||
totalFileSize: grp.packages.reduce((sum, p) => sum + p.fileSize, BigInt(0)).toString(),
|
totalFileSize: grp.packages.reduce((sum, p) => sum + p.fileSize, BigInt(0)).toString(),
|
||||||
totalFileCount: grp.packages.reduce((sum, p) => sum + p.fileCount, 0),
|
totalFileCount: grp.packages.reduce((sum, p) => sum + p.fileCount, 0),
|
||||||
packageCount: grp.packages.length,
|
packageCount: grp.packages.length,
|
||||||
@@ -227,7 +263,7 @@ export async function listDisplayItems(options: {
|
|||||||
archiveType: pkg.archiveType,
|
archiveType: pkg.archiveType,
|
||||||
fileCount: pkg.fileCount,
|
fileCount: pkg.fileCount,
|
||||||
isMultipart: pkg.isMultipart,
|
isMultipart: pkg.isMultipart,
|
||||||
hasPreview: pkg.previewData !== null,
|
hasPreview: packagePreviewIds.has(pkg.id),
|
||||||
creator: pkg.creator,
|
creator: pkg.creator,
|
||||||
tags: pkg.tags,
|
tags: pkg.tags,
|
||||||
indexedAt: pkg.indexedAt.toISOString(),
|
indexedAt: pkg.indexedAt.toISOString(),
|
||||||
@@ -440,13 +476,17 @@ export async function searchPackages(options: {
|
|||||||
indexedAt: true,
|
indexedAt: true,
|
||||||
creator: true,
|
creator: true,
|
||||||
tags: true,
|
tags: true,
|
||||||
previewData: true,
|
|
||||||
sourceChannel: { select: { id: true, title: true } },
|
sourceChannel: { select: { id: true, title: true } },
|
||||||
},
|
},
|
||||||
}),
|
}),
|
||||||
Promise.resolve(allIds.length),
|
Promise.resolve(allIds.length),
|
||||||
]);
|
]);
|
||||||
|
|
||||||
|
const previewIds = await fetchPreviewFlags(
|
||||||
|
"packages",
|
||||||
|
items.map((p) => p.id)
|
||||||
|
);
|
||||||
|
|
||||||
const mapped: PackageListItem[] = items.map((pkg) => ({
|
const mapped: PackageListItem[] = items.map((pkg) => ({
|
||||||
id: pkg.id,
|
id: pkg.id,
|
||||||
fileName: pkg.fileName,
|
fileName: pkg.fileName,
|
||||||
@@ -455,7 +495,7 @@ export async function searchPackages(options: {
|
|||||||
archiveType: pkg.archiveType,
|
archiveType: pkg.archiveType,
|
||||||
fileCount: pkg.fileCount,
|
fileCount: pkg.fileCount,
|
||||||
isMultipart: pkg.isMultipart,
|
isMultipart: pkg.isMultipart,
|
||||||
hasPreview: pkg.previewData !== null,
|
hasPreview: previewIds.has(pkg.id),
|
||||||
creator: pkg.creator,
|
creator: pkg.creator,
|
||||||
tags: pkg.tags,
|
tags: pkg.tags,
|
||||||
indexedAt: pkg.indexedAt.toISOString(),
|
indexedAt: pkg.indexedAt.toISOString(),
|
||||||
@@ -494,6 +534,15 @@ export async function getAllPackageTags(): Promise<string[]> {
|
|||||||
return result.map((r) => r.tag);
|
return result.map((r) => r.tag);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
export async function getAllPackageCreators(): Promise<string[]> {
|
||||||
|
const result = await prisma.$queryRaw<{ creator: string }[]>`
|
||||||
|
SELECT DISTINCT creator FROM packages
|
||||||
|
WHERE creator IS NOT NULL AND creator <> ''
|
||||||
|
ORDER BY creator
|
||||||
|
`;
|
||||||
|
return result.map((r) => r.creator);
|
||||||
|
}
|
||||||
|
|
||||||
export async function getIngestionStatus(): Promise<IngestionAccountStatus[]> {
|
export async function getIngestionStatus(): Promise<IngestionAccountStatus[]> {
|
||||||
const accounts = await prisma.telegramAccount.findMany({
|
const accounts = await prisma.telegramAccount.findMany({
|
||||||
orderBy: { createdAt: "asc" },
|
orderBy: { createdAt: "asc" },
|
||||||
@@ -550,6 +599,8 @@ export async function getIngestionStatus(): Promise<IngestionAccountStatus[]> {
|
|||||||
totalBytes: currentRun.totalBytes?.toString() ?? null,
|
totalBytes: currentRun.totalBytes?.toString() ?? null,
|
||||||
downloadPercent: currentRun.downloadPercent,
|
downloadPercent: currentRun.downloadPercent,
|
||||||
lastActivityAt: currentRun.lastActivityAt?.toISOString() ?? null,
|
lastActivityAt: currentRun.lastActivityAt?.toISOString() ?? null,
|
||||||
|
currentTopicId: currentRun.currentTopicId?.toString() ?? null,
|
||||||
|
currentAccountChannelMapId: currentRun.currentAccountChannelMapId,
|
||||||
}
|
}
|
||||||
: null,
|
: null,
|
||||||
});
|
});
|
||||||
@@ -634,13 +685,17 @@ export async function listUngroupedPackages(options: {
|
|||||||
partCount: true,
|
partCount: true,
|
||||||
tags: true,
|
tags: true,
|
||||||
indexedAt: true,
|
indexedAt: true,
|
||||||
previewData: true,
|
|
||||||
sourceChannel: { select: { id: true, title: true } },
|
sourceChannel: { select: { id: true, title: true } },
|
||||||
},
|
},
|
||||||
}),
|
}),
|
||||||
prisma.package.count({ where }),
|
prisma.package.count({ where }),
|
||||||
]);
|
]);
|
||||||
|
|
||||||
|
const previewIds = await fetchPreviewFlags(
|
||||||
|
"packages",
|
||||||
|
items.map((p) => p.id)
|
||||||
|
);
|
||||||
|
|
||||||
return {
|
return {
|
||||||
items: items.map((p) => ({
|
items: items.map((p) => ({
|
||||||
id: p.id,
|
id: p.id,
|
||||||
@@ -654,7 +709,7 @@ export async function listUngroupedPackages(options: {
|
|||||||
partCount: p.partCount,
|
partCount: p.partCount,
|
||||||
tags: p.tags,
|
tags: p.tags,
|
||||||
indexedAt: p.indexedAt.toISOString(),
|
indexedAt: p.indexedAt.toISOString(),
|
||||||
hasPreview: !!p.previewData,
|
hasPreview: previewIds.has(p.id),
|
||||||
sourceChannel: p.sourceChannel,
|
sourceChannel: p.sourceChannel,
|
||||||
matchedFileCount: 0,
|
matchedFileCount: 0,
|
||||||
matchedByContent: false,
|
matchedByContent: false,
|
||||||
|
|||||||
@@ -123,5 +123,7 @@ export interface IngestionAccountStatus {
|
|||||||
totalBytes: string | null; // BigInt serialized as string
|
totalBytes: string | null; // BigInt serialized as string
|
||||||
downloadPercent: number | null;
|
downloadPercent: number | null;
|
||||||
lastActivityAt: string | null;
|
lastActivityAt: string | null;
|
||||||
|
currentTopicId: string | null; // BigInt serialized as string
|
||||||
|
currentAccountChannelMapId: string | null;
|
||||||
} | null;
|
} | null;
|
||||||
}
|
}
|
||||||
|
|||||||
Generated
+1034
-1
File diff suppressed because it is too large
Load Diff
+5
-2
@@ -6,7 +6,9 @@
|
|||||||
"scripts": {
|
"scripts": {
|
||||||
"build": "tsc",
|
"build": "tsc",
|
||||||
"start": "node dist/index.js",
|
"start": "node dist/index.js",
|
||||||
"dev": "tsx watch src/index.ts"
|
"dev": "tsx watch src/index.ts",
|
||||||
|
"test": "vitest run",
|
||||||
|
"test:watch": "vitest"
|
||||||
},
|
},
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"@prisma/adapter-pg": "^7.4.0",
|
"@prisma/adapter-pg": "^7.4.0",
|
||||||
@@ -23,6 +25,7 @@
|
|||||||
"@types/yauzl": "^2.10.3",
|
"@types/yauzl": "^2.10.3",
|
||||||
"prisma": "^7.4.0",
|
"prisma": "^7.4.0",
|
||||||
"tsx": "^4.21.0",
|
"tsx": "^4.21.0",
|
||||||
"typescript": "^5"
|
"typescript": "^5",
|
||||||
|
"vitest": "^3.2.4"
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,69 @@
|
|||||||
|
import { describe, it, expect } from "vitest";
|
||||||
|
import { parseZipCentralDirectoryFromTail } from "./central-directory.js";
|
||||||
|
import { crc32 } from "zlib"; // Node 20+ exposes zlib.crc32
|
||||||
|
|
||||||
|
// Build a minimal STORE (no compression) ZIP in-memory with the given files.
|
||||||
|
function buildStoreZip(files: { name: string; data: Buffer }[]): Buffer {
|
||||||
|
const chunks: Buffer[] = [];
|
||||||
|
const central: Buffer[] = [];
|
||||||
|
let offset = 0;
|
||||||
|
for (const f of files) {
|
||||||
|
const crc = crc32(f.data) >>> 0;
|
||||||
|
const nameBuf = Buffer.from(f.name, "utf8");
|
||||||
|
const local = Buffer.alloc(30);
|
||||||
|
local.writeUInt32LE(0x04034b50, 0);
|
||||||
|
local.writeUInt16LE(20, 4); // version needed
|
||||||
|
local.writeUInt16LE(0, 6); // flags
|
||||||
|
local.writeUInt16LE(0, 8); // method = store
|
||||||
|
local.writeUInt32LE(crc, 14);
|
||||||
|
local.writeUInt32LE(f.data.length, 18); // compressed
|
||||||
|
local.writeUInt32LE(f.data.length, 22); // uncompressed
|
||||||
|
local.writeUInt16LE(nameBuf.length, 26);
|
||||||
|
local.writeUInt16LE(0, 28); // extra len
|
||||||
|
const localHeader = Buffer.concat([local, nameBuf, f.data]);
|
||||||
|
chunks.push(localHeader);
|
||||||
|
|
||||||
|
const cd = Buffer.alloc(46);
|
||||||
|
cd.writeUInt32LE(0x02014b50, 0);
|
||||||
|
cd.writeUInt16LE(20, 4); cd.writeUInt16LE(20, 6);
|
||||||
|
cd.writeUInt16LE(0, 8); cd.writeUInt16LE(0, 10);
|
||||||
|
cd.writeUInt32LE(crc, 16);
|
||||||
|
cd.writeUInt32LE(f.data.length, 20);
|
||||||
|
cd.writeUInt32LE(f.data.length, 24);
|
||||||
|
cd.writeUInt16LE(nameBuf.length, 28);
|
||||||
|
cd.writeUInt32LE(offset, 42); // local header offset
|
||||||
|
central.push(Buffer.concat([cd, nameBuf]));
|
||||||
|
offset += localHeader.length;
|
||||||
|
}
|
||||||
|
const cdBuf = Buffer.concat(central);
|
||||||
|
const cdOffset = offset;
|
||||||
|
const eocd = Buffer.alloc(22);
|
||||||
|
eocd.writeUInt32LE(0x06054b50, 0);
|
||||||
|
eocd.writeUInt16LE(files.length, 8);
|
||||||
|
eocd.writeUInt16LE(files.length, 10);
|
||||||
|
eocd.writeUInt32LE(cdBuf.length, 12);
|
||||||
|
eocd.writeUInt32LE(cdOffset, 16);
|
||||||
|
return Buffer.concat([...chunks, cdBuf, eocd]);
|
||||||
|
}
|
||||||
|
|
||||||
|
describe("parseZipCentralDirectoryFromTail", () => {
|
||||||
|
it("lists entries with correct names, sizes, and crc32", () => {
|
||||||
|
const zip = buildStoreZip([
|
||||||
|
{ name: "models/dragon.stl", data: Buffer.from("DRAGON") },
|
||||||
|
{ name: "readme.txt", data: Buffer.from("hello world") },
|
||||||
|
]);
|
||||||
|
const entries = parseZipCentralDirectoryFromTail(zip, 0);
|
||||||
|
expect(entries.map((e) => e.fileName).sort()).toEqual(["dragon.stl", "readme.txt"]);
|
||||||
|
const dragon = entries.find((e) => e.fileName === "dragon.stl")!;
|
||||||
|
expect(dragon.path).toBe("models/dragon.stl");
|
||||||
|
expect(dragon.uncompressedSize).toBe(6n);
|
||||||
|
expect(dragon.crc32).toMatch(/^[0-9a-f]{8}$/);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("throws when the central directory begins before the tail window", () => {
|
||||||
|
const zip = buildStoreZip([{ name: "a.txt", data: Buffer.alloc(100) }]);
|
||||||
|
// Provide only the last 30 bytes but claim they start at offset (len-30):
|
||||||
|
const tail = zip.subarray(zip.length - 30);
|
||||||
|
expect(() => parseZipCentralDirectoryFromTail(tail, zip.length - 30)).toThrow(RangeError);
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -0,0 +1,90 @@
|
|||||||
|
import path from "path";
|
||||||
|
import type { FileEntry } from "./zip-reader.js";
|
||||||
|
|
||||||
|
export const MIN_ZIP_TAIL_BYTES = 65_557;
|
||||||
|
|
||||||
|
const EOCD_SIG = 0x06054b50;
|
||||||
|
const CD_SIG = 0x02014b50;
|
||||||
|
|
||||||
|
function extOf(name: string): string | null {
|
||||||
|
const e = path.extname(name).replace(/^\./, "").toLowerCase();
|
||||||
|
return e === "" ? null : e;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Parse a ZIP central directory from the tail of an archive. */
|
||||||
|
export function parseZipCentralDirectoryFromTail(tail: Buffer, tailStart: number): FileEntry[] {
|
||||||
|
// 1. Find EOCD by scanning backward for its signature.
|
||||||
|
let eocd = -1;
|
||||||
|
for (let i = tail.length - 22; i >= 0; i--) {
|
||||||
|
if (tail.readUInt32LE(i) === EOCD_SIG) { eocd = i; break; }
|
||||||
|
}
|
||||||
|
if (eocd < 0) throw new RangeError("EOCD not found in tail");
|
||||||
|
|
||||||
|
let cdSize = tail.readUInt32LE(eocd + 12);
|
||||||
|
let cdOffset = tail.readUInt32LE(eocd + 16);
|
||||||
|
|
||||||
|
// ZIP64: sizes/offsets of 0xFFFFFFFF mean "see ZIP64 EOCD".
|
||||||
|
if (cdOffset === 0xffffffff || cdSize === 0xffffffff) {
|
||||||
|
const locSig = 0x07064b50;
|
||||||
|
let loc = -1;
|
||||||
|
for (let i = eocd - 20; i >= 0; i--) {
|
||||||
|
if (tail.readUInt32LE(i) === locSig) { loc = i; break; }
|
||||||
|
}
|
||||||
|
if (loc < 0) throw new RangeError("ZIP64 EOCD locator not in tail");
|
||||||
|
const z64Abs = Number(tail.readBigUInt64LE(loc + 8)); // absolute offset of ZIP64 EOCD
|
||||||
|
const z64 = z64Abs - tailStart;
|
||||||
|
if (z64 < 0) throw new RangeError("ZIP64 EOCD before tail window");
|
||||||
|
cdSize = Number(tail.readBigUInt64LE(z64 + 40));
|
||||||
|
cdOffset = Number(tail.readBigUInt64LE(z64 + 48));
|
||||||
|
}
|
||||||
|
|
||||||
|
// 2. Map the absolute central-directory offset into the tail buffer.
|
||||||
|
const cdLocal = cdOffset - tailStart;
|
||||||
|
if (cdLocal < 0 || cdLocal + cdSize > tail.length) {
|
||||||
|
throw new RangeError("Central directory begins before tail window");
|
||||||
|
}
|
||||||
|
|
||||||
|
// 3. Walk central-directory headers.
|
||||||
|
const entries: FileEntry[] = [];
|
||||||
|
let p = cdLocal;
|
||||||
|
const end = cdLocal + cdSize;
|
||||||
|
while (p + 46 <= end && tail.readUInt32LE(p) === CD_SIG) {
|
||||||
|
let crc = tail.readUInt32LE(p + 16) >>> 0;
|
||||||
|
let comp = BigInt(tail.readUInt32LE(p + 20));
|
||||||
|
let uncomp = BigInt(tail.readUInt32LE(p + 24));
|
||||||
|
const nameLen = tail.readUInt16LE(p + 28);
|
||||||
|
const extraLen = tail.readUInt16LE(p + 30);
|
||||||
|
const commentLen = tail.readUInt16LE(p + 32);
|
||||||
|
const name = tail.toString("utf8", p + 46, p + 46 + nameLen);
|
||||||
|
|
||||||
|
// ZIP64 extra field overrides 0xFFFFFFFF sizes.
|
||||||
|
if (comp === 0xffffffffn || uncomp === 0xffffffffn) {
|
||||||
|
let ep = p + 46 + nameLen;
|
||||||
|
const extraEnd = ep + extraLen;
|
||||||
|
while (ep + 4 <= extraEnd) {
|
||||||
|
const id = tail.readUInt16LE(ep);
|
||||||
|
const sz = tail.readUInt16LE(ep + 2);
|
||||||
|
if (id === 0x0001) {
|
||||||
|
let fp = ep + 4;
|
||||||
|
if (uncomp === 0xffffffffn) { uncomp = tail.readBigUInt64LE(fp); fp += 8; }
|
||||||
|
if (comp === 0xffffffffn) { comp = tail.readBigUInt64LE(fp); fp += 8; }
|
||||||
|
}
|
||||||
|
ep += 4 + sz;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const isDir = name.endsWith("/");
|
||||||
|
if (!isDir) {
|
||||||
|
entries.push({
|
||||||
|
path: name,
|
||||||
|
fileName: path.basename(name),
|
||||||
|
extension: extOf(name),
|
||||||
|
compressedSize: comp,
|
||||||
|
uncompressedSize: uncomp,
|
||||||
|
crc32: crc !== 0 ? crc.toString(16).padStart(8, "0") : null,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
p += 46 + nameLen + extraLen + commentLen;
|
||||||
|
}
|
||||||
|
return entries;
|
||||||
|
}
|
||||||
@@ -0,0 +1,35 @@
|
|||||||
|
import { describe, it, expect } from "vitest";
|
||||||
|
import { crcFingerprint, fingerprintsMatch } from "./fingerprint.js";
|
||||||
|
import type { FileEntry } from "./zip-reader.js";
|
||||||
|
|
||||||
|
const fe = (crc: string | null): FileEntry => ({
|
||||||
|
path: "a", fileName: "a", extension: null,
|
||||||
|
compressedSize: 0n, uncompressedSize: 0n, crc32: crc,
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("crcFingerprint", () => {
|
||||||
|
it("sorts crcs and marks complete", () => {
|
||||||
|
expect(crcFingerprint([fe("00ff"), fe("00aa")])).toEqual({ crcs: ["00aa", "00ff"], complete: true });
|
||||||
|
});
|
||||||
|
it("is incomplete when any crc is null", () => {
|
||||||
|
expect(crcFingerprint([fe("00aa"), fe(null)]).complete).toBe(false);
|
||||||
|
});
|
||||||
|
it("is incomplete when empty", () => {
|
||||||
|
expect(crcFingerprint([]).complete).toBe(false);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("fingerprintsMatch", () => {
|
||||||
|
it("matches identical crc multisets regardless of order", () => {
|
||||||
|
expect(fingerprintsMatch([fe("01"), fe("02")], [fe("02"), fe("01")])).toBe(true);
|
||||||
|
});
|
||||||
|
it("rejects different counts", () => {
|
||||||
|
expect(fingerprintsMatch([fe("01")], [fe("01"), fe("02")])).toBe(false);
|
||||||
|
});
|
||||||
|
it("rejects disjoint sets", () => {
|
||||||
|
expect(fingerprintsMatch([fe("01")], [fe("09")])).toBe(false);
|
||||||
|
});
|
||||||
|
it("rejects when either side is incomplete", () => {
|
||||||
|
expect(fingerprintsMatch([fe("01"), fe(null)], [fe("01"), fe("02")])).toBe(false);
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -0,0 +1,21 @@
|
|||||||
|
import type { FileEntry } from "./zip-reader.js";
|
||||||
|
|
||||||
|
export function crcFingerprint(entries: FileEntry[]): { crcs: string[]; complete: boolean } {
|
||||||
|
if (entries.length === 0) return { crcs: [], complete: false };
|
||||||
|
const crcs: string[] = [];
|
||||||
|
let complete = true;
|
||||||
|
for (const e of entries) {
|
||||||
|
if (e.crc32 == null) { complete = false; continue; }
|
||||||
|
crcs.push(e.crc32.toLowerCase());
|
||||||
|
}
|
||||||
|
crcs.sort();
|
||||||
|
return { crcs, complete };
|
||||||
|
}
|
||||||
|
|
||||||
|
export function fingerprintsMatch(a: FileEntry[], b: FileEntry[]): boolean {
|
||||||
|
const fa = crcFingerprint(a);
|
||||||
|
const fb = crcFingerprint(b);
|
||||||
|
if (!fa.complete || !fb.complete) return false;
|
||||||
|
if (fa.crcs.length !== fb.crcs.length) return false;
|
||||||
|
return fa.crcs.every((c, i) => c === fb.crcs[i]);
|
||||||
|
}
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
import { describe, it, expect, vi } from "vitest";
|
||||||
|
|
||||||
|
// logLevel is required here too (not just maxZipSizeMB/tempDir) because this
|
||||||
|
// mock replaces the config module for the whole test-file graph, including
|
||||||
|
// util/logger.ts's module-level `pino({ level: config.logLevel })` call —
|
||||||
|
// pino throws at import time if level is undefined.
|
||||||
|
vi.mock("../../util/config.js", () => ({ config: { maxZipSizeMB: 1, tempDir: "/tmp", logLevel: "info" } }));
|
||||||
|
const created: unknown[] = [];
|
||||||
|
vi.mock("../../db/client.js", () => ({
|
||||||
|
db: { systemNotification: { create: async (a: unknown) => { created.push(a); } } },
|
||||||
|
}));
|
||||||
|
|
||||||
|
import { fullDownloadListing } from "./fallback.js";
|
||||||
|
|
||||||
|
describe("fullDownloadListing", () => {
|
||||||
|
it("refuses to download over the size cap and records a notification", async () => {
|
||||||
|
const res = await fullDownloadListing({
|
||||||
|
client: {} as never,
|
||||||
|
parts: [{ fileId: "1", fileSize: 2n * 1024n * 1024n * 1024n, fileName: "big.rar" }],
|
||||||
|
archiveType: "RAR",
|
||||||
|
totalSize: 2n * 1024n * 1024n * 1024n,
|
||||||
|
fileName: "big.rar",
|
||||||
|
});
|
||||||
|
expect(res).toBeNull();
|
||||||
|
expect(created).toHaveLength(1);
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -0,0 +1,55 @@
|
|||||||
|
import { mkdtemp, rm } from "fs/promises";
|
||||||
|
import path from "path";
|
||||||
|
import type { Client } from "tdl";
|
||||||
|
import { config } from "../../util/config.js";
|
||||||
|
import { db } from "../../db/client.js";
|
||||||
|
import { childLogger } from "../../util/logger.js";
|
||||||
|
import { downloadFile } from "../../tdlib/download.js";
|
||||||
|
import { read7zContents } from "../sevenz-reader.js";
|
||||||
|
import { readRarContents } from "../rar-reader.js";
|
||||||
|
import type { FileEntry } from "../zip-reader.js";
|
||||||
|
import type { RangedPart } from "./sevenz-ranged.js";
|
||||||
|
|
||||||
|
const log = childLogger("ranged-fallback");
|
||||||
|
|
||||||
|
export async function fullDownloadListing(args: {
|
||||||
|
client: Client;
|
||||||
|
parts: RangedPart[];
|
||||||
|
archiveType: string;
|
||||||
|
totalSize: bigint;
|
||||||
|
fileName: string;
|
||||||
|
}): Promise<FileEntry[] | null> {
|
||||||
|
const capBytes = BigInt(config.maxZipSizeMB) * 1024n * 1024n;
|
||||||
|
if (args.totalSize > capBytes) {
|
||||||
|
await db.systemNotification.create({
|
||||||
|
data: {
|
||||||
|
type: "INTEGRITY_AUDIT",
|
||||||
|
severity: "WARNING",
|
||||||
|
title: `Listing skipped (over size cap): ${args.fileName}`,
|
||||||
|
message: `Ranged listing failed and the archive (${args.totalSize} bytes) exceeds WORKER_MAX_ZIP_SIZE_MB; not downloaded. Inner files left unindexed.`,
|
||||||
|
context: { fileName: args.fileName, archiveType: args.archiveType },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
log.warn({ fileName: args.fileName }, "fallback skipped — over size cap");
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
const dir = await mkdtemp(path.join(config.tempDir, "fallback-"));
|
||||||
|
const paths: string[] = [];
|
||||||
|
try {
|
||||||
|
for (const p of args.parts) {
|
||||||
|
const dest = path.join(dir, p.fileName);
|
||||||
|
await downloadFile(args.client, p.fileId, dest, p.fileSize, p.fileName, () => {});
|
||||||
|
paths.push(dest);
|
||||||
|
}
|
||||||
|
const entries =
|
||||||
|
args.archiveType === "SEVEN_Z" ? await read7zContents(paths[0])
|
||||||
|
: args.archiveType === "RAR" ? await readRarContents(paths[0])
|
||||||
|
: [];
|
||||||
|
return entries.length > 0 ? entries : null;
|
||||||
|
} catch (err) {
|
||||||
|
log.warn({ err, fileName: args.fileName }, "full-download fallback failed");
|
||||||
|
return null;
|
||||||
|
} finally {
|
||||||
|
await rm(dir, { recursive: true, force: true });
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,14 @@
|
|||||||
|
import type { Client } from "tdl";
|
||||||
|
import { downloadFileRange } from "../../tdlib/range-download.js";
|
||||||
|
|
||||||
|
export type RangeReader = (
|
||||||
|
fileId: string,
|
||||||
|
offset: number,
|
||||||
|
length: number,
|
||||||
|
partSize: bigint,
|
||||||
|
) => Promise<Buffer>;
|
||||||
|
|
||||||
|
export function tdlibRangeReader(client: Client): RangeReader {
|
||||||
|
return (fileId, offset, length, partSize) =>
|
||||||
|
downloadFileRange(client, fileId, offset, length, partSize);
|
||||||
|
}
|
||||||
@@ -0,0 +1,150 @@
|
|||||||
|
import { describe, it, expect } from "vitest";
|
||||||
|
import { readVint, detectRarSignature, parseRar5BlockExtent, parseRar4BlockExtent, walkRarVolume, readRarListingRanged } from "./rar-ranged.js";
|
||||||
|
import type { RangeReader } from "./range-reader.js";
|
||||||
|
|
||||||
|
describe("readVint", () => {
|
||||||
|
it("reads single-byte and multi-byte values (base-128 LE)", () => {
|
||||||
|
expect(readVint(Buffer.from([0x08]), 0)).toEqual({ value: 8, bytes: 1 });
|
||||||
|
// 0x80,0x01 => 0 | (1<<7) = 128
|
||||||
|
expect(readVint(Buffer.from([0x80, 0x01]), 0)).toEqual({ value: 128, bytes: 2 });
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("detectRarSignature", () => {
|
||||||
|
it("detects RAR5 and RAR4", () => {
|
||||||
|
expect(detectRarSignature(Buffer.from([0x52,0x61,0x72,0x21,0x1a,0x07,0x01,0x00]))).toEqual({ version: 5, sigLen: 8 });
|
||||||
|
expect(detectRarSignature(Buffer.from([0x52,0x61,0x72,0x21,0x1a,0x07,0x00]))).toEqual({ version: 4, sigLen: 7 });
|
||||||
|
expect(detectRarSignature(Buffer.alloc(8))).toBeNull();
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("parseRar5BlockExtent", () => {
|
||||||
|
it("computes header+data extent and flags end-of-archive", () => {
|
||||||
|
// CRC32(4) | HeaderSize vint=5 | Type vint=2 (file) | Flags vint=2 (data present) | DataSize vint=100 | (pad to headerSize)
|
||||||
|
const b = Buffer.concat([
|
||||||
|
Buffer.from([0,0,0,0]), // CRC
|
||||||
|
Buffer.from([0x05]), // HeaderSize = 5 (bytes after this vint)
|
||||||
|
Buffer.from([0x02]), // Type = 2 (file)
|
||||||
|
Buffer.from([0x02]), // Flags = 0x02 -> data present
|
||||||
|
Buffer.from([0x64]), // DataSize = 100
|
||||||
|
Buffer.from([0x00, 0x00]), // padding to fill HeaderSize(5): Type+Flags+DataSize=3, +2 pad =5
|
||||||
|
]);
|
||||||
|
const ext = parseRar5BlockExtent(b, 0);
|
||||||
|
// headerBytes = 4 (CRC) + 1 (HeaderSize vint) + 5 (HeaderSize) = 10
|
||||||
|
expect(ext.headerBytes).toBe(10);
|
||||||
|
expect(ext.dataSize).toBe(100);
|
||||||
|
expect(ext.isEnd).toBe(false);
|
||||||
|
|
||||||
|
const endBlk = Buffer.from([0,0,0,0, 0x02, 0x05, 0x00]); // HeaderSize=2, Type=5(end), Flags=0
|
||||||
|
const e2 = parseRar5BlockExtent(endBlk, 0);
|
||||||
|
expect(e2.isEnd).toBe(true);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("parseRar4BlockExtent", () => {
|
||||||
|
it("computes extent with ADD_SIZE when flag 0x8000 is set", () => {
|
||||||
|
// CRC(2) TYPE(1)=0x74 FLAGS(2)=0x8000 HEAD_SIZE(2)=11 ADD_SIZE(4)=200
|
||||||
|
const b = Buffer.alloc(11);
|
||||||
|
b.writeUInt8(0x74, 2);
|
||||||
|
b.writeUInt16LE(0x8000, 3);
|
||||||
|
b.writeUInt16LE(11, 5);
|
||||||
|
b.writeUInt32LE(200, 7);
|
||||||
|
const ext = parseRar4BlockExtent(b, 0);
|
||||||
|
expect(ext.headerBytes).toBe(11);
|
||||||
|
expect(ext.dataSize).toBe(200);
|
||||||
|
expect(ext.isEnd).toBe(false);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
// Build a synthetic RAR5 volume: signature + main header + 2 file blocks (each
|
||||||
|
// with data) + end block. We only need extents to be walkable.
|
||||||
|
function buildRar5Volume(): Buffer {
|
||||||
|
const sig = Buffer.from([0x52,0x61,0x72,0x21,0x1a,0x07,0x01,0x00]);
|
||||||
|
const block = (type: number, flags: number, dataSize: number, pad = 0) => {
|
||||||
|
const body = [Buffer.from([type]), Buffer.from([flags])];
|
||||||
|
if (flags & 0x0002) body.push(Buffer.from([dataSize])); // DataSize (<=127 for test)
|
||||||
|
if (pad) body.push(Buffer.alloc(pad));
|
||||||
|
const bodyBuf = Buffer.concat(body);
|
||||||
|
const hs = Buffer.from([bodyBuf.length]); // HeaderSize vint (<=127)
|
||||||
|
const header = Buffer.concat([Buffer.alloc(4), hs, bodyBuf]); // CRC(4)+HeaderSize+body
|
||||||
|
const data = Buffer.alloc(flags & 0x0002 ? dataSize : 0, 0xEE);
|
||||||
|
return Buffer.concat([header, data]);
|
||||||
|
};
|
||||||
|
const main = block(1, 0, 0); // main archive header, no data
|
||||||
|
const f1 = block(2, 0x02, 20); // file header + 20 bytes data
|
||||||
|
const f2 = block(2, 0x02, 30); // file header + 30 bytes data
|
||||||
|
const end = block(5, 0, 0); // end of archive
|
||||||
|
return Buffer.concat([sig, main, f1, f2, end]);
|
||||||
|
}
|
||||||
|
|
||||||
|
describe("walkRarVolume", () => {
|
||||||
|
it("harvests every block header and stops at end-of-archive", async () => {
|
||||||
|
const vol = buildRar5Volume();
|
||||||
|
const read: RangeReader = async (_id, offset, length) => vol.subarray(offset, offset + length);
|
||||||
|
const regions = await walkRarVolume(read, { fileId: "1", fileSize: BigInt(vol.length), fileName: "a.rar" }, 5, 8);
|
||||||
|
expect(regions).not.toBeNull();
|
||||||
|
// main + 2 files + end = 4 header regions
|
||||||
|
expect(regions!).toHaveLength(4);
|
||||||
|
// First region starts right after the 8-byte signature
|
||||||
|
expect(regions![0].offset).toBe(8);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("returns null when a block claims an absurd header size (corrupt/desynced)", async () => {
|
||||||
|
// RAR5 block with HeaderSize vint encoding a value > 8MB.
|
||||||
|
// Encode 9_000_000 as RAR vint: bytes little-endian 7-bit groups with continuation bit.
|
||||||
|
function encodeVint(n: number): number[] {
|
||||||
|
const out: number[] = [];
|
||||||
|
while (n >= 0x80) {
|
||||||
|
out.push((n & 0x7f) | 0x80);
|
||||||
|
n = Math.floor(n / 128);
|
||||||
|
}
|
||||||
|
out.push(n);
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
const sig = Buffer.from([0x52, 0x61, 0x72, 0x21, 0x1a, 0x07, 0x01, 0x00]); // RAR5 signature
|
||||||
|
const hsVint = encodeVint(9_000_000);
|
||||||
|
// Block = CRC(4) + HeaderSize vint(9MB) + Type(1 byte) + Flags(1 byte)
|
||||||
|
const block = Buffer.concat([Buffer.alloc(4), Buffer.from(hsVint), Buffer.from([0x02, 0x00])]);
|
||||||
|
const vol = Buffer.concat([sig, block]);
|
||||||
|
const size = 20 * 1024 * 1024;
|
||||||
|
const read = async (_id: string, offset: number, length: number) => {
|
||||||
|
if (offset >= vol.length) return Buffer.alloc(0);
|
||||||
|
return vol.subarray(offset, Math.min(offset + length, vol.length));
|
||||||
|
};
|
||||||
|
const regions = await walkRarVolume(read, { fileId: "1", fileSize: BigInt(size), fileName: "c.rar" }, 5, 8);
|
||||||
|
expect(regions).toBeNull();
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("readRarListingRanged (single part)", () => {
|
||||||
|
it("returns null cleanly when the reconstructed file isn't a real RAR", async () => {
|
||||||
|
const vol = buildRar5Volume();
|
||||||
|
const read: RangeReader = async (_id, offset, length) => vol.subarray(offset, offset + length);
|
||||||
|
const res = await readRarListingRanged([{ fileId: "1", fileSize: BigInt(vol.length), fileName: "a.rar" }], read);
|
||||||
|
expect(res === null || Array.isArray(res)).toBe(true); // real unrar parse covered live
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("readRarListingRanged (multipart)", () => {
|
||||||
|
it("walks each volume from its own signature and reconstructs all parts", async () => {
|
||||||
|
const vol = buildRar5Volume(); // reuse from Task 6 test
|
||||||
|
// Two volumes with identical structure; each RangeReader read is scoped by fileId.
|
||||||
|
const byId: Record<string, Buffer> = { p1: vol, p2: vol };
|
||||||
|
const reads: Record<string, number> = { p1: 0, p2: 0 };
|
||||||
|
const read: RangeReader = async (fileId, offset, length) => {
|
||||||
|
reads[fileId]++;
|
||||||
|
return byId[fileId].subarray(offset, offset + length);
|
||||||
|
};
|
||||||
|
const res = await readRarListingRanged(
|
||||||
|
[
|
||||||
|
{ fileId: "p1", fileSize: BigInt(vol.length), fileName: "x.part1.rar" },
|
||||||
|
{ fileId: "p2", fileSize: BigInt(vol.length), fileName: "x.part2.rar" },
|
||||||
|
],
|
||||||
|
read,
|
||||||
|
);
|
||||||
|
// Both volumes were walked (each read at least its signature + blocks).
|
||||||
|
expect(reads.p1).toBeGreaterThan(0);
|
||||||
|
expect(reads.p2).toBeGreaterThan(0);
|
||||||
|
expect(res === null || Array.isArray(res)).toBe(true);
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -0,0 +1,110 @@
|
|||||||
|
const RAR4_SIG = Buffer.from([0x52, 0x61, 0x72, 0x21, 0x1a, 0x07, 0x00]);
|
||||||
|
const RAR5_SIG = Buffer.from([0x52, 0x61, 0x72, 0x21, 0x1a, 0x07, 0x01, 0x00]);
|
||||||
|
|
||||||
|
export function readVint(buf: Buffer, pos: number): { value: number; bytes: number } {
|
||||||
|
let value = 0, shift = 0, bytes = 0;
|
||||||
|
while (pos + bytes < buf.length) {
|
||||||
|
const b = buf[pos + bytes];
|
||||||
|
value += (b & 0x7f) * Math.pow(2, shift); // Math.pow keeps >32-bit sizes exact up to 2^53
|
||||||
|
bytes++;
|
||||||
|
if ((b & 0x80) === 0) return { value, bytes };
|
||||||
|
shift += 7;
|
||||||
|
if (shift > 63) break;
|
||||||
|
}
|
||||||
|
throw new RangeError("incomplete RAR vint");
|
||||||
|
}
|
||||||
|
|
||||||
|
export function detectRarSignature(buf: Buffer): { version: 4 | 5; sigLen: number } | null {
|
||||||
|
if (buf.length >= 8 && buf.subarray(0, 8).equals(RAR5_SIG)) return { version: 5, sigLen: 8 };
|
||||||
|
if (buf.length >= 7 && buf.subarray(0, 7).equals(RAR4_SIG)) return { version: 4, sigLen: 7 };
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface BlockExtent { headerBytes: number; dataSize: number; isEnd: boolean }
|
||||||
|
|
||||||
|
// RAR5: CRC32(4) | HeaderSize(vint) | HeaderType(vint) | HeaderFlags(vint)
|
||||||
|
// [ExtraAreaSize(vint) if flags&0x0001] [DataSize(vint) if flags&0x0002] ...
|
||||||
|
export function parseRar5BlockExtent(buf: Buffer, pos: number): BlockExtent {
|
||||||
|
let p = pos + 4; // skip CRC32
|
||||||
|
const hs = readVint(buf, p); p += hs.bytes;
|
||||||
|
const headerBytes = 4 + hs.bytes + hs.value; // CRC + HeaderSize-vint + HeaderSize
|
||||||
|
const type = readVint(buf, p); p += type.bytes;
|
||||||
|
const flags = readVint(buf, p); p += flags.bytes;
|
||||||
|
if (flags.value & 0x0001) { const ea = readVint(buf, p); p += ea.bytes; } // extra area size (skip)
|
||||||
|
let dataSize = 0;
|
||||||
|
if (flags.value & 0x0002) { const ds = readVint(buf, p); p += ds.bytes; dataSize = ds.value; }
|
||||||
|
return { headerBytes, dataSize, isEnd: type.value === 5 };
|
||||||
|
}
|
||||||
|
|
||||||
|
// RAR4: HEAD_CRC(2) | HEAD_TYPE(1) | HEAD_FLAGS(2) | HEAD_SIZE(2) [ADD_SIZE(4) if flags&0x8000]
|
||||||
|
export function parseRar4BlockExtent(buf: Buffer, pos: number): BlockExtent {
|
||||||
|
const type = buf.readUInt8(pos + 2);
|
||||||
|
const flags = buf.readUInt16LE(pos + 3);
|
||||||
|
const headSize = buf.readUInt16LE(pos + 5);
|
||||||
|
const dataSize = (flags & 0x8000) ? buf.readUInt32LE(pos + 7) : 0;
|
||||||
|
return { headerBytes: headSize, dataSize, isEnd: type === 0x7b };
|
||||||
|
}
|
||||||
|
|
||||||
|
import type { FileEntry } from "../zip-reader.js";
|
||||||
|
import type { RangeReader } from "./range-reader.js";
|
||||||
|
import type { RangedPart } from "./sevenz-ranged.js";
|
||||||
|
import { listFromSparse, type SparsePart } from "./sparse-list.js";
|
||||||
|
import { readRarContents } from "../rar-reader.js";
|
||||||
|
import { childLogger } from "../../util/logger.js";
|
||||||
|
|
||||||
|
const rlog = childLogger("rar-ranged");
|
||||||
|
const MAX_RAR_BLOCKS = 50000;
|
||||||
|
const MAX_RAR_HEADER_BYTES = 8 * 1024 * 1024; // 8 MB — real RAR block headers are far smaller; guards against a corrupt/desynced HeaderSize
|
||||||
|
const HEADER_CHUNK = 8192;
|
||||||
|
|
||||||
|
export async function walkRarVolume(
|
||||||
|
read: RangeReader,
|
||||||
|
part: RangedPart,
|
||||||
|
version: 4 | 5,
|
||||||
|
sigLen: number,
|
||||||
|
): Promise<{ offset: number; bytes: Buffer }[] | null> {
|
||||||
|
const size = Number(part.fileSize);
|
||||||
|
const regions: { offset: number; bytes: Buffer }[] = [];
|
||||||
|
let pos = sigLen;
|
||||||
|
let blocks = 0;
|
||||||
|
try {
|
||||||
|
while (pos < size) {
|
||||||
|
if (++blocks > MAX_RAR_BLOCKS) return null;
|
||||||
|
const chunkLen = Math.min(HEADER_CHUNK, size - pos);
|
||||||
|
let chunk = await read(part.fileId, pos, chunkLen, part.fileSize);
|
||||||
|
const ext = version === 5 ? parseRar5BlockExtent(chunk, 0) : parseRar4BlockExtent(chunk, 0);
|
||||||
|
if (ext.headerBytes > MAX_RAR_HEADER_BYTES) return null;
|
||||||
|
// Ensure we have the full header bytes to harvest (long filenames).
|
||||||
|
let headerBuf = chunk;
|
||||||
|
if (ext.headerBytes > chunk.length) {
|
||||||
|
headerBuf = await read(part.fileId, pos, Math.min(ext.headerBytes, size - pos), part.fileSize);
|
||||||
|
}
|
||||||
|
regions.push({ offset: pos, bytes: headerBuf.subarray(0, Math.min(ext.headerBytes, size - pos)) });
|
||||||
|
if (ext.isEnd) break;
|
||||||
|
const advance = ext.headerBytes + ext.dataSize;
|
||||||
|
if (advance <= 0) return null;
|
||||||
|
if (pos + advance > size) break; // data clamped at the volume boundary (multipart continuation)
|
||||||
|
pos += advance;
|
||||||
|
}
|
||||||
|
return regions;
|
||||||
|
} catch (err) {
|
||||||
|
rlog.warn({ err, fileId: part.fileId }, "RAR volume walk failed");
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
export async function readRarListingRanged(
|
||||||
|
parts: RangedPart[],
|
||||||
|
read: RangeReader,
|
||||||
|
): Promise<FileEntry[] | null> {
|
||||||
|
const sparseParts: SparsePart[] = [];
|
||||||
|
for (const part of parts) {
|
||||||
|
const head = await read(part.fileId, 0, 16, part.fileSize);
|
||||||
|
const sig = detectRarSignature(head);
|
||||||
|
if (!sig) return null;
|
||||||
|
const regions = await walkRarVolume(read, part, sig.version, sig.sigLen);
|
||||||
|
if (!regions) return null;
|
||||||
|
sparseParts.push({ fileName: part.fileName, size: Number(part.fileSize), regions });
|
||||||
|
}
|
||||||
|
return listFromSparse(sparseParts, readRarContents);
|
||||||
|
}
|
||||||
@@ -0,0 +1,133 @@
|
|||||||
|
import { describe, it, expect } from "vitest";
|
||||||
|
import { parseSevenZSignatureHeader } from "./sevenz-ranged.js";
|
||||||
|
|
||||||
|
const MAGIC = Buffer.from([0x37, 0x7a, 0xbc, 0xaf, 0x27, 0x1c]);
|
||||||
|
|
||||||
|
function buildSignatureHeader(nextOffset: bigint, nextSize: bigint): Buffer {
|
||||||
|
const buf = Buffer.alloc(32);
|
||||||
|
MAGIC.copy(buf, 0);
|
||||||
|
buf.writeUInt8(0, 6); buf.writeUInt8(4, 7); // version 0.4
|
||||||
|
buf.writeUInt32LE(0, 8); // StartHeaderCRC (unused here)
|
||||||
|
buf.writeBigUInt64LE(nextOffset, 12);
|
||||||
|
buf.writeBigUInt64LE(nextSize, 20);
|
||||||
|
buf.writeUInt32LE(0, 28); // NextHeaderCRC (unused here)
|
||||||
|
return buf;
|
||||||
|
}
|
||||||
|
|
||||||
|
describe("parseSevenZSignatureHeader", () => {
|
||||||
|
it("reads NextHeaderOffset and NextHeaderSize", () => {
|
||||||
|
const buf = buildSignatureHeader(1_000_000n, 4096n);
|
||||||
|
expect(parseSevenZSignatureHeader(buf)).toEqual({ nextHeaderOffset: 1_000_000, nextHeaderSize: 4096 });
|
||||||
|
});
|
||||||
|
|
||||||
|
it("returns null on bad magic", () => {
|
||||||
|
expect(parseSevenZSignatureHeader(Buffer.alloc(32))).toBeNull();
|
||||||
|
});
|
||||||
|
|
||||||
|
it("returns null when shorter than 32 bytes", () => {
|
||||||
|
expect(parseSevenZSignatureHeader(MAGIC)).toBeNull();
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
import { readSevenZListingRanged } from "./sevenz-ranged.js";
|
||||||
|
import type { RangeReader } from "./range-reader.js";
|
||||||
|
|
||||||
|
describe("readSevenZListingRanged", () => {
|
||||||
|
it("reads the signature + end-header regions and reconstructs for 7z l", async () => {
|
||||||
|
const size = 5_000_000;
|
||||||
|
const endHeaderOffset = 4_900_000; // absolute
|
||||||
|
const nextHeaderOffset = endHeaderOffset - 32;
|
||||||
|
const sig = Buffer.alloc(32);
|
||||||
|
Buffer.from([0x37, 0x7a, 0xbc, 0xaf, 0x27, 0x1c]).copy(sig, 0);
|
||||||
|
sig.writeBigUInt64LE(BigInt(nextHeaderOffset), 12);
|
||||||
|
sig.writeBigUInt64LE(100n, 20);
|
||||||
|
|
||||||
|
const reads: { offset: number; length: number }[] = [];
|
||||||
|
const read: RangeReader = async (_id, offset, length) => {
|
||||||
|
reads.push({ offset, length });
|
||||||
|
if (offset === 0) return sig.subarray(0, length);
|
||||||
|
return Buffer.alloc(length, 0xAB); // stand-in end-header bytes
|
||||||
|
};
|
||||||
|
|
||||||
|
// Inject a fake lister via the module boundary: readSevenZListingRanged
|
||||||
|
// calls listFromSparse(parts, read7zContents). We assert the ranged reads
|
||||||
|
// it issued; the sparse file + real 7z is covered by live verification.
|
||||||
|
const entries = await readSevenZListingRanged(
|
||||||
|
[{ fileId: "1", fileSize: BigInt(size), fileName: "a.7z" }],
|
||||||
|
read,
|
||||||
|
);
|
||||||
|
// entries may be null here because the stand-in bytes aren't a real 7z;
|
||||||
|
// the contract under test is the ranged-read offsets:
|
||||||
|
expect(reads[0]).toEqual({ offset: 0, length: 32 });
|
||||||
|
expect(reads[1]).toEqual({ offset: endHeaderOffset, length: 100 });
|
||||||
|
expect(entries === null || Array.isArray(entries)).toBe(true);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
import { read7zNumber, locate7zEncodedHeaderPack } from "./sevenz-ranged.js";
|
||||||
|
|
||||||
|
describe("read7zNumber", () => {
|
||||||
|
it("reads a single-byte number", () => {
|
||||||
|
expect(read7zNumber(Buffer.from([0x2a]), 0)).toEqual({ value: 42, next: 1 });
|
||||||
|
});
|
||||||
|
it("reads a two-byte number (first-byte length mask + LE trailing byte)", () => {
|
||||||
|
// 1000 = 0x03E8: low byte 0xE8, high nibble 0x03 -> first = 0x80|0x03 = 0x83, trailing 0xE8
|
||||||
|
expect(read7zNumber(Buffer.from([0x83, 0xe8]), 0)).toEqual({ value: 1000, next: 2 });
|
||||||
|
// 500 = 0x01F4 -> first 0x81, trailing 0xF4
|
||||||
|
expect(read7zNumber(Buffer.from([0x81, 0xf4]), 0)).toEqual({ value: 500, next: 2 });
|
||||||
|
});
|
||||||
|
it("throws when pos starts past the buffer end (short read)", () => {
|
||||||
|
expect(() => read7zNumber(Buffer.from([0x2a]), 5)).toThrow(RangeError);
|
||||||
|
expect(() => read7zNumber(Buffer.alloc(0), 0)).toThrow(RangeError);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("locate7zEncodedHeaderPack", () => {
|
||||||
|
it("parses PackPos and summed PackSize from an encoded header", () => {
|
||||||
|
// kEncodedHeader, kPackInfo, PackPos=1000([0x83,0xe8]), NumStreams=1([0x01]),
|
||||||
|
// kSize, PackSize=500([0x81,0xf4])
|
||||||
|
const enc = Buffer.from([0x17, 0x06, 0x83, 0xe8, 0x01, 0x09, 0x81, 0xf4]);
|
||||||
|
expect(locate7zEncodedHeaderPack(enc)).toEqual({ packPos: 1000, packSize: 500 });
|
||||||
|
});
|
||||||
|
it("returns null for a plain (kHeader 0x01) header", () => {
|
||||||
|
expect(locate7zEncodedHeaderPack(Buffer.from([0x01, 0x04]))).toBeNull();
|
||||||
|
});
|
||||||
|
it("sums multiple pack streams", () => {
|
||||||
|
// PackPos=0([0x00]), NumStreams=2([0x02]), kSize, sizes 10([0x0a]) + 20([0x14])
|
||||||
|
const enc = Buffer.from([0x17, 0x06, 0x00, 0x02, 0x09, 0x0a, 0x14]);
|
||||||
|
expect(locate7zEncodedHeaderPack(enc)).toEqual({ packPos: 0, packSize: 30 });
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("readSevenZListingRanged (encoded header)", () => {
|
||||||
|
it("fetches the mid-file packed-header region as a third read", async () => {
|
||||||
|
const size = 5_000_000;
|
||||||
|
const nextHeaderOffset = 4_000_000; // relative to end of 32-byte sig header
|
||||||
|
const endStart = 32 + nextHeaderOffset; // absolute
|
||||||
|
const packPos = 1000; // relative to end of sig header
|
||||||
|
const packStart = 32 + packPos; // absolute
|
||||||
|
const packSize = 500;
|
||||||
|
const sig = Buffer.alloc(32);
|
||||||
|
Buffer.from([0x37, 0x7a, 0xbc, 0xaf, 0x27, 0x1c]).copy(sig, 0);
|
||||||
|
sig.writeBigUInt64LE(BigInt(nextHeaderOffset), 12);
|
||||||
|
sig.writeBigUInt64LE(8n, 20); // NextHeaderSize = 8 (the encoded-header descriptor below)
|
||||||
|
const encHeader = Buffer.from([0x17, 0x06, 0x83, 0xe8, 0x01, 0x09, 0x81, 0xf4]);
|
||||||
|
|
||||||
|
const reads: { offset: number; length: number }[] = [];
|
||||||
|
const read = async (_id: string, offset: number, length: number) => {
|
||||||
|
reads.push({ offset, length });
|
||||||
|
if (offset === 0) return sig.subarray(0, length);
|
||||||
|
if (offset === endStart) return encHeader.subarray(0, length);
|
||||||
|
return Buffer.alloc(length, 0xcd); // stand-in packed-header bytes
|
||||||
|
};
|
||||||
|
|
||||||
|
const entries = await readSevenZListingRanged(
|
||||||
|
[{ fileId: "1", fileSize: BigInt(size), fileName: "a.7z" }],
|
||||||
|
read,
|
||||||
|
);
|
||||||
|
expect(reads[0]).toEqual({ offset: 0, length: 32 });
|
||||||
|
expect(reads[1]).toEqual({ offset: endStart, length: 8 });
|
||||||
|
expect(reads[2]).toEqual({ offset: packStart, length: packSize });
|
||||||
|
expect(entries === null || Array.isArray(entries)).toBe(true);
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -0,0 +1,112 @@
|
|||||||
|
import type { FileEntry } from "../zip-reader.js";
|
||||||
|
import { read7zContents } from "../sevenz-reader.js";
|
||||||
|
import { listFromSparse } from "./sparse-list.js";
|
||||||
|
import type { RangeReader } from "./range-reader.js";
|
||||||
|
import { childLogger } from "../../util/logger.js";
|
||||||
|
|
||||||
|
const log = childLogger("sevenz-ranged");
|
||||||
|
|
||||||
|
const SEVENZ_MAGIC = Buffer.from([0x37, 0x7a, 0xbc, 0xaf, 0x27, 0x1c]);
|
||||||
|
|
||||||
|
const K_HEADER = 0x01;
|
||||||
|
const K_ENCODED_HEADER = 0x17;
|
||||||
|
const K_PACK_INFO = 0x06;
|
||||||
|
const K_SIZE = 0x09;
|
||||||
|
|
||||||
|
/** Read a 7z variable-length number: first byte is a length mask, followed by
|
||||||
|
* little-endian bytes. Math.pow keeps values exact above 2^31. */
|
||||||
|
export function read7zNumber(buf: Buffer, pos: number): { value: number; next: number } {
|
||||||
|
if (pos >= buf.length) throw new RangeError("7z number reads past buffer end");
|
||||||
|
const first = buf[pos];
|
||||||
|
let mask = 0x80;
|
||||||
|
let value = 0;
|
||||||
|
let p = pos + 1;
|
||||||
|
for (let i = 0; i < 8; i++) {
|
||||||
|
if ((first & mask) === 0) {
|
||||||
|
value += (first & (mask - 1)) * Math.pow(2, 8 * i);
|
||||||
|
return { value, next: p };
|
||||||
|
}
|
||||||
|
if (p >= buf.length) throw new RangeError("7z number overruns buffer");
|
||||||
|
value += buf[p] * Math.pow(2, 8 * i);
|
||||||
|
p++;
|
||||||
|
mask >>= 1;
|
||||||
|
}
|
||||||
|
return { value, next: p };
|
||||||
|
}
|
||||||
|
|
||||||
|
/** For an encoded (kEncodedHeader) 7z next-header, return the absolute-ish
|
||||||
|
* location of the packed header stream(s): PackPos (relative to end of the
|
||||||
|
* 32-byte signature header) and the summed PackSize. Null if not encoded or
|
||||||
|
* the StreamsInfo doesn't start with PackInfo as expected. */
|
||||||
|
export function locate7zEncodedHeaderPack(
|
||||||
|
nextHeader: Buffer,
|
||||||
|
): { packPos: number; packSize: number } | null {
|
||||||
|
let p = 0;
|
||||||
|
if (nextHeader[p] !== K_ENCODED_HEADER) return null;
|
||||||
|
p++;
|
||||||
|
if (nextHeader[p] !== K_PACK_INFO) return null;
|
||||||
|
p++;
|
||||||
|
const packPos = read7zNumber(nextHeader, p); p = packPos.next;
|
||||||
|
const numStreams = read7zNumber(nextHeader, p); p = numStreams.next;
|
||||||
|
if (nextHeader[p] !== K_SIZE) return null;
|
||||||
|
p++;
|
||||||
|
let total = 0;
|
||||||
|
for (let i = 0; i < numStreams.value; i++) {
|
||||||
|
const s = read7zNumber(nextHeader, p); p = s.next;
|
||||||
|
total += s.value;
|
||||||
|
}
|
||||||
|
return { packPos: packPos.value, packSize: total };
|
||||||
|
}
|
||||||
|
|
||||||
|
export function parseSevenZSignatureHeader(
|
||||||
|
buf: Buffer,
|
||||||
|
): { nextHeaderOffset: number; nextHeaderSize: number } | null {
|
||||||
|
if (buf.length < 32) return null;
|
||||||
|
if (!buf.subarray(0, 6).equals(SEVENZ_MAGIC)) return null;
|
||||||
|
return {
|
||||||
|
nextHeaderOffset: Number(buf.readBigUInt64LE(12)),
|
||||||
|
nextHeaderSize: Number(buf.readBigUInt64LE(20)),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface RangedPart { fileId: string; fileSize: bigint; fileName: string }
|
||||||
|
|
||||||
|
export async function readSevenZListingRanged(
|
||||||
|
parts: RangedPart[],
|
||||||
|
read: RangeReader,
|
||||||
|
): Promise<FileEntry[] | null> {
|
||||||
|
const part = parts[0];
|
||||||
|
if (!part) return null;
|
||||||
|
const size = Number(part.fileSize);
|
||||||
|
try {
|
||||||
|
const sig = await read(part.fileId, 0, 32, part.fileSize);
|
||||||
|
const parsed = parseSevenZSignatureHeader(sig);
|
||||||
|
if (!parsed) return null;
|
||||||
|
const endStart = 32 + parsed.nextHeaderOffset;
|
||||||
|
if (endStart < 0 || endStart + parsed.nextHeaderSize > size) return null;
|
||||||
|
const endHeader = await read(part.fileId, endStart, parsed.nextHeaderSize, part.fileSize);
|
||||||
|
|
||||||
|
const regions = [
|
||||||
|
{ offset: 0, bytes: sig },
|
||||||
|
{ offset: endStart, bytes: endHeader },
|
||||||
|
];
|
||||||
|
|
||||||
|
const headerType = endHeader[0];
|
||||||
|
if (headerType === K_ENCODED_HEADER) {
|
||||||
|
// Compressed header: its packed bytes live mid-file, not at EOF. Fetch them.
|
||||||
|
const pack = locate7zEncodedHeaderPack(endHeader);
|
||||||
|
if (!pack) return null;
|
||||||
|
const packStart = 32 + pack.packPos;
|
||||||
|
if (packStart < 0 || packStart + pack.packSize > size) return null;
|
||||||
|
const packBytes = await read(part.fileId, packStart, pack.packSize, part.fileSize);
|
||||||
|
regions.push({ offset: packStart, bytes: packBytes });
|
||||||
|
} else if (headerType !== K_HEADER) {
|
||||||
|
return null; // unknown next-header type
|
||||||
|
}
|
||||||
|
|
||||||
|
return listFromSparse([{ fileName: part.fileName, size, regions }], read7zContents);
|
||||||
|
} catch (err) {
|
||||||
|
log.warn({ err, fileId: part.fileId }, "ranged 7z listing failed");
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,41 @@
|
|||||||
|
import { describe, it, expect } from "vitest";
|
||||||
|
import { open } from "fs/promises";
|
||||||
|
import { listFromSparse } from "./sparse-list.js";
|
||||||
|
|
||||||
|
describe("listFromSparse", () => {
|
||||||
|
it("writes each region at its offset into a sparse file and passes the path to the lister", async () => {
|
||||||
|
const size = 1_000_000;
|
||||||
|
const regions = [
|
||||||
|
{ offset: 0, bytes: Buffer.from("HEAD") },
|
||||||
|
{ offset: size - 4, bytes: Buffer.from("TAIL") },
|
||||||
|
];
|
||||||
|
let seenPath = "";
|
||||||
|
const entries = await listFromSparse(
|
||||||
|
[{ fileName: "sample.7z", size, regions }],
|
||||||
|
async (firstPartPath) => {
|
||||||
|
seenPath = firstPartPath;
|
||||||
|
const fh = await open(firstPartPath, "r");
|
||||||
|
try {
|
||||||
|
const head = Buffer.alloc(4); await fh.read(head, 0, 4, 0);
|
||||||
|
const tail = Buffer.alloc(4); await fh.read(tail, 0, 4, size - 4);
|
||||||
|
const hole = Buffer.alloc(4); await fh.read(hole, 0, 4, 500_000);
|
||||||
|
expect(head.toString()).toBe("HEAD");
|
||||||
|
expect(tail.toString()).toBe("TAIL");
|
||||||
|
expect(hole.equals(Buffer.alloc(4))).toBe(true); // gap is zero
|
||||||
|
} finally { await fh.close(); }
|
||||||
|
return [{ path: "a/b.stl", fileName: "b.stl", extension: "stl", compressedSize: 1n, uncompressedSize: 1n, crc32: null }];
|
||||||
|
},
|
||||||
|
);
|
||||||
|
expect(seenPath.endsWith("sample.7z")).toBe(true);
|
||||||
|
expect(entries).not.toBeNull();
|
||||||
|
expect(entries!).toHaveLength(1);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("returns null when the lister yields no entries", async () => {
|
||||||
|
const res = await listFromSparse(
|
||||||
|
[{ fileName: "x.7z", size: 100, regions: [{ offset: 0, bytes: Buffer.from("A") }] }],
|
||||||
|
async () => [],
|
||||||
|
);
|
||||||
|
expect(res).toBeNull();
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -0,0 +1,52 @@
|
|||||||
|
import { mkdtemp, open, rm } from "fs/promises";
|
||||||
|
import path from "path";
|
||||||
|
import { config } from "../../util/config.js";
|
||||||
|
import { childLogger } from "../../util/logger.js";
|
||||||
|
import type { FileEntry } from "../zip-reader.js";
|
||||||
|
|
||||||
|
const log = childLogger("sparse-list");
|
||||||
|
|
||||||
|
export interface SparsePart {
|
||||||
|
fileName: string;
|
||||||
|
size: number;
|
||||||
|
regions: { offset: number; bytes: Buffer }[];
|
||||||
|
}
|
||||||
|
|
||||||
|
export type SparseLister = (firstPartPath: string) => Promise<FileEntry[]>;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Reconstruct archive header bytes into sparse temp files (data areas left as
|
||||||
|
* zero holes), run `lister` on the first part, return its entries.
|
||||||
|
* Returns null on any error or when the lister finds nothing.
|
||||||
|
*/
|
||||||
|
export async function listFromSparse(
|
||||||
|
parts: SparsePart[],
|
||||||
|
lister: SparseLister,
|
||||||
|
): Promise<FileEntry[] | null> {
|
||||||
|
if (parts.length === 0) return null;
|
||||||
|
const dir = await mkdtemp(path.join(config.tempDir, "ranged-"));
|
||||||
|
try {
|
||||||
|
let firstPath = "";
|
||||||
|
for (let i = 0; i < parts.length; i++) {
|
||||||
|
const p = parts[i];
|
||||||
|
const filePath = path.join(dir, p.fileName);
|
||||||
|
if (i === 0) firstPath = filePath;
|
||||||
|
const fh = await open(filePath, "w");
|
||||||
|
try {
|
||||||
|
await fh.truncate(p.size); // create the sparse hole
|
||||||
|
for (const r of p.regions) {
|
||||||
|
await fh.write(r.bytes, 0, r.bytes.length, r.offset);
|
||||||
|
}
|
||||||
|
} finally {
|
||||||
|
await fh.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const entries = await lister(firstPath);
|
||||||
|
return entries.length > 0 ? entries : null;
|
||||||
|
} catch (err) {
|
||||||
|
log.warn({ err }, "sparse listing failed");
|
||||||
|
return null;
|
||||||
|
} finally {
|
||||||
|
await rm(dir, { recursive: true, force: true });
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,5 +1,6 @@
|
|||||||
import { db } from "./client.js";
|
import { db } from "./client.js";
|
||||||
import type { ArchiveType, FetchStatus } from "@prisma/client";
|
import type { ArchiveType, FetchStatus } from "@prisma/client";
|
||||||
|
import type { FileEntry } from "../archive/zip-reader.js";
|
||||||
|
|
||||||
export async function getActiveAccounts() {
|
export async function getActiveAccounts() {
|
||||||
return db.telegramAccount.findMany({
|
return db.telegramAccount.findMany({
|
||||||
@@ -369,10 +370,13 @@ export interface ActivityUpdate {
|
|||||||
downloadedBytes?: bigint | null;
|
downloadedBytes?: bigint | null;
|
||||||
totalBytes?: bigint | null;
|
totalBytes?: bigint | null;
|
||||||
downloadPercent?: number | null;
|
downloadPercent?: number | null;
|
||||||
|
currentTopicId?: bigint | null;
|
||||||
|
currentAccountChannelMapId?: string | null;
|
||||||
messagesScanned?: number;
|
messagesScanned?: number;
|
||||||
zipsFound?: number;
|
zipsFound?: number;
|
||||||
zipsDuplicate?: number;
|
zipsDuplicate?: number;
|
||||||
zipsIngested?: number;
|
zipsIngested?: number;
|
||||||
|
zipsBackfilled?: number;
|
||||||
}
|
}
|
||||||
|
|
||||||
export async function updateRunActivity(
|
export async function updateRunActivity(
|
||||||
@@ -396,6 +400,11 @@ export async function updateRunActivity(
|
|||||||
...(activity.zipsFound !== undefined && { zipsFound: activity.zipsFound }),
|
...(activity.zipsFound !== undefined && { zipsFound: activity.zipsFound }),
|
||||||
...(activity.zipsDuplicate !== undefined && { zipsDuplicate: activity.zipsDuplicate }),
|
...(activity.zipsDuplicate !== undefined && { zipsDuplicate: activity.zipsDuplicate }),
|
||||||
...(activity.zipsIngested !== undefined && { zipsIngested: activity.zipsIngested }),
|
...(activity.zipsIngested !== undefined && { zipsIngested: activity.zipsIngested }),
|
||||||
|
...(activity.zipsBackfilled !== undefined && { zipsBackfilled: activity.zipsBackfilled }),
|
||||||
|
...(activity.currentTopicId !== undefined && { currentTopicId: activity.currentTopicId }),
|
||||||
|
...(activity.currentAccountChannelMapId !== undefined && {
|
||||||
|
currentAccountChannelMapId: activity.currentAccountChannelMapId,
|
||||||
|
}),
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
@@ -410,6 +419,8 @@ const CLEAR_ACTIVITY = {
|
|||||||
downloadedBytes: null,
|
downloadedBytes: null,
|
||||||
totalBytes: null,
|
totalBytes: null,
|
||||||
downloadPercent: null,
|
downloadPercent: null,
|
||||||
|
currentTopicId: null,
|
||||||
|
currentAccountChannelMapId: null,
|
||||||
lastActivityAt: new Date(),
|
lastActivityAt: new Date(),
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -420,6 +431,7 @@ export async function completeIngestionRun(
|
|||||||
zipsFound: number;
|
zipsFound: number;
|
||||||
zipsDuplicate: number;
|
zipsDuplicate: number;
|
||||||
zipsIngested: number;
|
zipsIngested: number;
|
||||||
|
zipsBackfilled: number;
|
||||||
}
|
}
|
||||||
) {
|
) {
|
||||||
return db.ingestionRun.update({
|
return db.ingestionRun.update({
|
||||||
@@ -997,3 +1009,142 @@ export async function createAutoGroup(input: {
|
|||||||
|
|
||||||
return group.id;
|
return group.id;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ── Provenance backfill ──
|
||||||
|
|
||||||
|
export interface PlaceholderCandidate {
|
||||||
|
id: string;
|
||||||
|
archiveType: string;
|
||||||
|
fileName: string;
|
||||||
|
fileCount: number;
|
||||||
|
fileSize: bigint;
|
||||||
|
destMessageId: bigint | null;
|
||||||
|
destMessageIds: bigint[];
|
||||||
|
destChannel: { telegramId: bigint } | null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Find every placeholder Package matching name+size (oldest first). Package
|
||||||
|
* has no direct `destChannel` relation (only the scalar `destChannelId`), so
|
||||||
|
* each row's destination TelegramChannel telegramId is resolved with a
|
||||||
|
* follow-up lookup rather than a Prisma include.
|
||||||
|
*/
|
||||||
|
export async function findPlaceholderCandidates(
|
||||||
|
destChannelId: string,
|
||||||
|
fileName: string,
|
||||||
|
fileSize: bigint,
|
||||||
|
): Promise<PlaceholderCandidate[]> {
|
||||||
|
const rows = await db.package.findMany({
|
||||||
|
where: {
|
||||||
|
fileName,
|
||||||
|
fileSize,
|
||||||
|
destMessageId: { not: null },
|
||||||
|
// Placeholder provenance (spec §1): manual-upload (source == destination)
|
||||||
|
// OR rebuild record (sourceMessageId == 0 "unknown" sentinel).
|
||||||
|
OR: [
|
||||||
|
{ sourceChannelId: destChannelId },
|
||||||
|
{ sourceMessageId: 0n },
|
||||||
|
],
|
||||||
|
},
|
||||||
|
select: {
|
||||||
|
id: true, archiveType: true, fileName: true, fileCount: true, fileSize: true,
|
||||||
|
destMessageId: true, destMessageIds: true, destChannelId: true,
|
||||||
|
},
|
||||||
|
orderBy: { indexedAt: "asc" },
|
||||||
|
});
|
||||||
|
if (rows.length === 0) return [];
|
||||||
|
|
||||||
|
const destChannelIds = [...new Set(rows.map((r) => r.destChannelId).filter((id): id is string => !!id))];
|
||||||
|
const channels = destChannelIds.length
|
||||||
|
? await db.telegramChannel.findMany({
|
||||||
|
where: { id: { in: destChannelIds } },
|
||||||
|
select: { id: true, telegramId: true },
|
||||||
|
})
|
||||||
|
: [];
|
||||||
|
const telegramIdById = new Map(channels.map((c) => [c.id, c.telegramId]));
|
||||||
|
|
||||||
|
return rows.map((row) => ({
|
||||||
|
id: row.id,
|
||||||
|
archiveType: row.archiveType,
|
||||||
|
fileName: row.fileName,
|
||||||
|
fileCount: row.fileCount,
|
||||||
|
fileSize: row.fileSize,
|
||||||
|
destMessageId: row.destMessageId,
|
||||||
|
destMessageIds: row.destMessageIds,
|
||||||
|
destChannel: row.destChannelId && telegramIdById.has(row.destChannelId)
|
||||||
|
? { telegramId: telegramIdById.get(row.destChannelId)! }
|
||||||
|
: null,
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
export async function findPlaceholderCandidate(
|
||||||
|
destChannelId: string,
|
||||||
|
fileName: string,
|
||||||
|
fileSize: bigint,
|
||||||
|
): Promise<PlaceholderCandidate | null> {
|
||||||
|
return (await findPlaceholderCandidates(destChannelId, fileName, fileSize))[0] ?? null;
|
||||||
|
}
|
||||||
|
|
||||||
|
export async function getPackageFileCrcs(packageId: string): Promise<(string | null)[]> {
|
||||||
|
const rows = await db.packageFile.findMany({
|
||||||
|
where: { packageId },
|
||||||
|
select: { crc32: true },
|
||||||
|
});
|
||||||
|
return rows.map((r) => r.crc32);
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface BackfillProvenanceInput {
|
||||||
|
packageId: string;
|
||||||
|
destChannelId: string; // to re-check placeholder status in-txn
|
||||||
|
sourceChannelId: string;
|
||||||
|
sourceMessageId: bigint;
|
||||||
|
sourceTopicId: bigint | null;
|
||||||
|
sourceCaption: string | null;
|
||||||
|
remoteUniqueId: string | null;
|
||||||
|
creator: string | null; // always set (re-derived by caller)
|
||||||
|
entries?: FileEntry[]; // set only if candidate had fileCount === 0
|
||||||
|
previewData?: Buffer | null; // set only if provided and candidate lacks one
|
||||||
|
previewMsgId?: bigint | null;
|
||||||
|
}
|
||||||
|
|
||||||
|
export async function backfillProvenance(input: BackfillProvenanceInput): Promise<boolean> {
|
||||||
|
return db.$transaction(async (tx) => {
|
||||||
|
const current = await tx.package.findUnique({
|
||||||
|
where: { id: input.packageId },
|
||||||
|
select: { sourceChannelId: true, sourceMessageId: true, previewData: true, fileCount: true },
|
||||||
|
});
|
||||||
|
// Re-check placeholder status inside the txn (another worker may have won).
|
||||||
|
// Placeholder = manual-upload (source==dest) OR rebuild (sourceMessageId==0).
|
||||||
|
const stillPlaceholder =
|
||||||
|
!!current &&
|
||||||
|
(current.sourceChannelId === input.destChannelId || current.sourceMessageId === 0n);
|
||||||
|
if (!stillPlaceholder) return false;
|
||||||
|
|
||||||
|
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||||
|
const data: any = {
|
||||||
|
sourceChannelId: input.sourceChannelId,
|
||||||
|
sourceMessageId: input.sourceMessageId,
|
||||||
|
sourceTopicId: input.sourceTopicId,
|
||||||
|
sourceCaption: input.sourceCaption,
|
||||||
|
remoteUniqueId: input.remoteUniqueId,
|
||||||
|
creator: input.creator,
|
||||||
|
};
|
||||||
|
if (input.entries && current.fileCount === 0) {
|
||||||
|
await tx.packageFile.deleteMany({ where: { packageId: input.packageId } });
|
||||||
|
await tx.packageFile.createMany({
|
||||||
|
data: input.entries.map((e) => ({
|
||||||
|
packageId: input.packageId,
|
||||||
|
path: e.path, fileName: e.fileName, extension: e.extension,
|
||||||
|
compressedSize: e.compressedSize, uncompressedSize: e.uncompressedSize, crc32: e.crc32,
|
||||||
|
})),
|
||||||
|
});
|
||||||
|
data.fileCount = input.entries.length;
|
||||||
|
}
|
||||||
|
if (input.previewData && !current.previewData) {
|
||||||
|
data.previewData = input.previewData;
|
||||||
|
data.previewMsgId = input.previewMsgId ?? null;
|
||||||
|
}
|
||||||
|
await tx.package.update({ where: { id: input.packageId }, data });
|
||||||
|
return true;
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,306 @@
|
|||||||
|
import { db } from "./db/client.js";
|
||||||
|
import { childLogger } from "./util/logger.js";
|
||||||
|
import { downloadFileRange } from "./tdlib/range-download.js";
|
||||||
|
import { invokeWithTimeout } from "./tdlib/download.js";
|
||||||
|
import { parseZipCentralDirectoryFromTail, MIN_ZIP_TAIL_BYTES } from "./archive/central-directory.js";
|
||||||
|
import { fingerprintsMatch, crcFingerprint } from "./archive/fingerprint.js";
|
||||||
|
import {
|
||||||
|
findPlaceholderCandidates,
|
||||||
|
getPackageFileCrcs,
|
||||||
|
backfillProvenance,
|
||||||
|
type PlaceholderCandidate,
|
||||||
|
} from "./db/queries.js";
|
||||||
|
import type { FileEntry } from "./archive/zip-reader.js";
|
||||||
|
import { readSevenZListingRanged, type RangedPart } from "./archive/ranged/sevenz-ranged.js";
|
||||||
|
import { readRarListingRanged } from "./archive/ranged/rar-ranged.js";
|
||||||
|
import { tdlibRangeReader } from "./archive/ranged/range-reader.js";
|
||||||
|
import { fullDownloadListing } from "./archive/ranged/fallback.js";
|
||||||
|
import type { Client } from "tdl";
|
||||||
|
|
||||||
|
const log = childLogger("provenance-backfill");
|
||||||
|
|
||||||
|
export interface BackfillArgs {
|
||||||
|
client: Client;
|
||||||
|
destChannelId: string;
|
||||||
|
scannedSourceChannelId: string;
|
||||||
|
fileName: string;
|
||||||
|
fileSize: bigint;
|
||||||
|
archiveType: string;
|
||||||
|
sourceMessageId: bigint;
|
||||||
|
sourceTopicId: bigint | null;
|
||||||
|
sourceCaption: string | null;
|
||||||
|
remoteUniqueId: string | null;
|
||||||
|
creator: string | null;
|
||||||
|
scannedParts: RangedPart[];
|
||||||
|
previewData?: Buffer | null;
|
||||||
|
previewMsgId?: bigint | null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Read a ZIP central directory from the tail of a (possibly multipart)
|
||||||
|
* archive. `parts` is ordered; only the LAST part carries the EOCD record.
|
||||||
|
* `fileSize` on each part is that part's own size (NOT the whole-archive
|
||||||
|
* total) so the download offset stays within that part's bounds, while
|
||||||
|
* `tailStart` passed to the parser is the logical whole-archive offset
|
||||||
|
* (preceding parts' sizes + the offset within the last part).
|
||||||
|
*/
|
||||||
|
async function readScannedZipListing(
|
||||||
|
client: Client,
|
||||||
|
parts: { fileId: string; fileSize: bigint }[],
|
||||||
|
): Promise<FileEntry[] | null> {
|
||||||
|
if (parts.length === 0) return null;
|
||||||
|
const lastPart = parts[parts.length - 1];
|
||||||
|
const precedingSize = parts.slice(0, -1).reduce((sum, p) => sum + Number(p.fileSize), 0);
|
||||||
|
const lastSize = Number(lastPart.fileSize);
|
||||||
|
for (const tailBytes of [MIN_ZIP_TAIL_BYTES, MIN_ZIP_TAIL_BYTES * 4]) {
|
||||||
|
const partOffset = Math.max(0, lastSize - tailBytes);
|
||||||
|
const downloadLen = Math.min(tailBytes, lastSize);
|
||||||
|
try {
|
||||||
|
const buf = await downloadFileRange(client, lastPart.fileId, partOffset, downloadLen, lastPart.fileSize);
|
||||||
|
const tailStart = precedingSize + partOffset;
|
||||||
|
return parseZipCentralDirectoryFromTail(buf, tailStart);
|
||||||
|
} catch (err) {
|
||||||
|
if (err instanceof RangeError) continue; // try a larger tail
|
||||||
|
log.warn({ err, fileId: lastPart.fileId }, "ranged ZIP listing failed");
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Resolve the destination copy's message(s) into ranged parts (file id +
|
||||||
|
* size + name), in order, so a multipart destination copy is reconstructed
|
||||||
|
* with correct per-part sizes and names (the last message carries the
|
||||||
|
* EOCD-bearing tail part for ZIP; multipart RAR needs correctly-suffixed
|
||||||
|
* `.partN.rar` names for `unrar` sibling discovery). Cheap-only: any TDLib
|
||||||
|
* failure here degrades the caller to name-size confidence rather than
|
||||||
|
* falling back to a full download.
|
||||||
|
*/
|
||||||
|
async function resolveDestParts(
|
||||||
|
client: Client,
|
||||||
|
destChatTelegramId: bigint,
|
||||||
|
destMessageIds: bigint[],
|
||||||
|
destMessageId: bigint | null,
|
||||||
|
fallbackFileName: string,
|
||||||
|
): Promise<RangedPart[] | null> {
|
||||||
|
const messageIds = destMessageIds.length > 0 ? destMessageIds : destMessageId ? [destMessageId] : [];
|
||||||
|
if (messageIds.length === 0) return null;
|
||||||
|
try {
|
||||||
|
const parts: RangedPart[] = [];
|
||||||
|
for (const msgId of messageIds) {
|
||||||
|
const msg = (await invokeWithTimeout(client, {
|
||||||
|
_: "getMessage",
|
||||||
|
chat_id: Number(destChatTelegramId),
|
||||||
|
message_id: Number(msgId),
|
||||||
|
})) as { content?: { document?: { document?: { id: number; size?: number }; file_name?: string } } };
|
||||||
|
const doc = msg?.content?.document?.document;
|
||||||
|
if (!doc?.id) return null;
|
||||||
|
const fileName = msg?.content?.document?.file_name || fallbackFileName;
|
||||||
|
parts.push({ fileId: String(doc.id), fileSize: BigInt(doc.size ?? 0), fileName });
|
||||||
|
}
|
||||||
|
return parts;
|
||||||
|
} catch (err) {
|
||||||
|
log.warn({ err, destMessageIds: messageIds.map(Number) }, "destination archive part resolution failed");
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async function readScannedListingRanged(
|
||||||
|
archiveType: string,
|
||||||
|
client: Client,
|
||||||
|
parts: RangedPart[],
|
||||||
|
): Promise<FileEntry[] | null> {
|
||||||
|
const read = tdlibRangeReader(client);
|
||||||
|
if (archiveType === "ZIP") return readScannedZipListing(client, parts);
|
||||||
|
if (archiveType === "SEVEN_Z") return readSevenZListingRanged(parts, read);
|
||||||
|
if (archiveType === "RAR") return readRarListingRanged(parts, read);
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Build the CRC fingerprint entries for a placeholder candidate: start from
|
||||||
|
* its stored PackageFile CRCs, and if those are incomplete (e.g. a rebuild
|
||||||
|
* candidate with fileCount === 0), fall back to a fresh ranged read of the
|
||||||
|
* candidate's own copy in the destination channel (Task 9).
|
||||||
|
*/
|
||||||
|
async function resolveCandidateFingerprintEntries(
|
||||||
|
client: Client,
|
||||||
|
candidate: PlaceholderCandidate,
|
||||||
|
): Promise<FileEntry[]> {
|
||||||
|
const candidateCrcs = await getPackageFileCrcs(candidate.id);
|
||||||
|
let candidateEntries: FileEntry[] = candidateCrcs.map((crc) => ({
|
||||||
|
path: "", fileName: "", extension: null, compressedSize: 0n, uncompressedSize: 0n, crc32: crc,
|
||||||
|
}));
|
||||||
|
const hasDestMessage = candidate.destMessageIds.length > 0 || candidate.destMessageId != null;
|
||||||
|
if (!crcFingerprint(candidateEntries).complete && hasDestMessage && candidate.destChannel) {
|
||||||
|
const destParts = await resolveDestParts(
|
||||||
|
client,
|
||||||
|
candidate.destChannel.telegramId,
|
||||||
|
candidate.destMessageIds,
|
||||||
|
candidate.destMessageId,
|
||||||
|
candidate.fileName,
|
||||||
|
);
|
||||||
|
let destEntries: FileEntry[] | null = null;
|
||||||
|
if (destParts) {
|
||||||
|
const read = tdlibRangeReader(client);
|
||||||
|
destEntries =
|
||||||
|
candidate.archiveType === "ZIP" ? await readScannedZipListing(client, destParts)
|
||||||
|
: candidate.archiveType === "SEVEN_Z" ? await readSevenZListingRanged(destParts, read)
|
||||||
|
: candidate.archiveType === "RAR" ? await readRarListingRanged(destParts, read)
|
||||||
|
: null;
|
||||||
|
}
|
||||||
|
if (destEntries) {
|
||||||
|
candidateEntries = destEntries;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return candidateEntries;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Classify a fingerprint comparison between two entry sets. "incomplete"
|
||||||
|
* means at least one side is missing CRCs (e.g. an empty file → CRC32 of
|
||||||
|
* zero-length data → null) and the comparison CANNOT be used to confirm or
|
||||||
|
* refute a match — callers must fall back to name+size confidence rather
|
||||||
|
* than treating this as a mismatch.
|
||||||
|
*/
|
||||||
|
function compareFingerprints(a: FileEntry[], b: FileEntry[]): "match" | "mismatch" | "incomplete" {
|
||||||
|
const fa = crcFingerprint(a);
|
||||||
|
const fb = crcFingerprint(b);
|
||||||
|
if (!fa.complete || !fb.complete) return "incomplete";
|
||||||
|
return fingerprintsMatch(a, b) ? "match" : "mismatch";
|
||||||
|
}
|
||||||
|
|
||||||
|
export async function tryProvenanceBackfill(
|
||||||
|
args: BackfillArgs,
|
||||||
|
): Promise<{ backfilled: boolean; confidence?: "fingerprint" | "name-size" }> {
|
||||||
|
const candidates = await findPlaceholderCandidates(args.destChannelId, args.fileName, args.fileSize);
|
||||||
|
if (candidates.length === 0) return { backfilled: false };
|
||||||
|
|
||||||
|
let scannedEntries: FileEntry[] | null = await readScannedListingRanged(
|
||||||
|
args.archiveType, args.client, args.scannedParts,
|
||||||
|
);
|
||||||
|
// Cheap ranged read failed — fall back to a size-capped full download so the
|
||||||
|
// listing still gets indexed. Only worth it when the candidate lacks a listing.
|
||||||
|
if (!scannedEntries && candidates.some((c) => c.fileCount === 0)) {
|
||||||
|
const totalSize = args.scannedParts.reduce((s, p) => s + p.fileSize, 0n);
|
||||||
|
scannedEntries = await fullDownloadListing({
|
||||||
|
client: args.client, parts: args.scannedParts, archiveType: args.archiveType,
|
||||||
|
totalSize, fileName: args.fileName,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
let chosen = candidates[0];
|
||||||
|
let confidence: "fingerprint" | "name-size" = "name-size";
|
||||||
|
|
||||||
|
if (candidates.length > 1) {
|
||||||
|
// Multiple placeholder packages share this name+size. Try to
|
||||||
|
// disambiguate by fingerprint (ZIP only); if we can't uniquely resolve
|
||||||
|
// it, notify instead of guessing which one is the real match.
|
||||||
|
if (args.archiveType === "ZIP" && scannedEntries) {
|
||||||
|
const matches: PlaceholderCandidate[] = [];
|
||||||
|
// Candidates NOT ruled out as a definite (both-complete) mismatch —
|
||||||
|
// used as the name+size fallback pool when the fingerprint can't
|
||||||
|
// confirm a match (e.g. incomplete CRCs on either side).
|
||||||
|
const nonMismatches: PlaceholderCandidate[] = [];
|
||||||
|
for (const c of candidates) {
|
||||||
|
const candidateEntries = await resolveCandidateFingerprintEntries(args.client, c);
|
||||||
|
const comparison = compareFingerprints(scannedEntries, candidateEntries);
|
||||||
|
if (comparison === "match") {
|
||||||
|
matches.push(c);
|
||||||
|
nonMismatches.push(c);
|
||||||
|
} else if (comparison === "incomplete") {
|
||||||
|
nonMismatches.push(c);
|
||||||
|
}
|
||||||
|
// comparison === "mismatch": both sides complete and differ — excluded.
|
||||||
|
}
|
||||||
|
if (matches.length === 1) {
|
||||||
|
chosen = matches[0];
|
||||||
|
confidence = "fingerprint";
|
||||||
|
} else if (matches.length === 0 && nonMismatches.length === 1) {
|
||||||
|
// Fingerprint couldn't confirm (incomplete CRCs), but exactly one
|
||||||
|
// candidate wasn't ruled out as a definite mismatch — fall back to
|
||||||
|
// name+size confidence rather than treating this as unresolved.
|
||||||
|
chosen = nonMismatches[0];
|
||||||
|
confidence = "name-size";
|
||||||
|
} else {
|
||||||
|
await db.systemNotification.create({
|
||||||
|
data: {
|
||||||
|
type: "INTEGRITY_AUDIT",
|
||||||
|
severity: "WARNING",
|
||||||
|
title: `Ambiguous provenance match: ${args.fileName}`,
|
||||||
|
message: `${candidates.length} placeholder packages share this name+size and the fingerprint did not uniquely disambiguate. No provenance was backfilled.`,
|
||||||
|
context: { fileName: args.fileName, candidateIds: candidates.map((c) => c.id) },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
return { backfilled: false };
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
// Can't disambiguate without a fingerprint — notify, don't guess.
|
||||||
|
await db.systemNotification.create({
|
||||||
|
data: {
|
||||||
|
type: "INTEGRITY_AUDIT",
|
||||||
|
severity: "WARNING",
|
||||||
|
title: `Ambiguous provenance match: ${args.fileName}`,
|
||||||
|
message: `${candidates.length} placeholder packages share this name+size (archive type ${args.archiveType} — no cheap fingerprint). No provenance was backfilled.`,
|
||||||
|
context: { fileName: args.fileName, candidateIds: candidates.map((c) => c.id) },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
return { backfilled: false };
|
||||||
|
}
|
||||||
|
} else if (scannedEntries) {
|
||||||
|
const candidateEntries = await resolveCandidateFingerprintEntries(args.client, chosen);
|
||||||
|
const comparison = compareFingerprints(scannedEntries, candidateEntries);
|
||||||
|
if (comparison === "match") {
|
||||||
|
confidence = "fingerprint";
|
||||||
|
} else if (comparison === "mismatch") {
|
||||||
|
// Both sides' CRCs are complete and differ: NOT the same content
|
||||||
|
// despite name+size. Do not backfill.
|
||||||
|
log.info({ candidateId: chosen.id, fileName: args.fileName }, "fingerprint mismatch — not backfilling");
|
||||||
|
return { backfilled: false };
|
||||||
|
}
|
||||||
|
// comparison === "incomplete": can't confirm or refute by fingerprint —
|
||||||
|
// fall through and backfill on name+size confidence instead.
|
||||||
|
}
|
||||||
|
|
||||||
|
const ok = await backfillProvenance({
|
||||||
|
packageId: chosen.id,
|
||||||
|
destChannelId: args.destChannelId,
|
||||||
|
sourceChannelId: args.scannedSourceChannelId,
|
||||||
|
sourceMessageId: args.sourceMessageId,
|
||||||
|
sourceTopicId: args.sourceTopicId,
|
||||||
|
sourceCaption: args.sourceCaption,
|
||||||
|
remoteUniqueId: args.remoteUniqueId,
|
||||||
|
creator: args.creator,
|
||||||
|
entries: chosen.fileCount === 0 && scannedEntries ? scannedEntries : undefined,
|
||||||
|
previewData: args.previewData ?? undefined,
|
||||||
|
previewMsgId: args.previewMsgId ?? undefined,
|
||||||
|
});
|
||||||
|
|
||||||
|
if (!ok) return { backfilled: false };
|
||||||
|
|
||||||
|
if (confidence === "name-size") {
|
||||||
|
// Lower-confidence backfill: no CRC fingerprint guard confirmed this
|
||||||
|
// match. Record it as an auditable event so name+size-only backfills
|
||||||
|
// can be reviewed after the fact.
|
||||||
|
await db.systemNotification.create({
|
||||||
|
data: {
|
||||||
|
type: "INTEGRITY_AUDIT",
|
||||||
|
severity: "INFO",
|
||||||
|
title: `Provenance backfilled by name+size: ${args.fileName}`,
|
||||||
|
message: `Package ${chosen.id} was matched to a scanned source message by file name and size only (no CRC fingerprint confirmation).`,
|
||||||
|
context: {
|
||||||
|
packageId: chosen.id,
|
||||||
|
fileName: args.fileName,
|
||||||
|
sourceChannelId: args.scannedSourceChannelId,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
log.info(
|
||||||
|
{ candidateId: chosen.id, fileName: args.fileName, confidence, source: args.scannedSourceChannelId },
|
||||||
|
"provenance backfilled",
|
||||||
|
);
|
||||||
|
return { backfilled: true, confidence };
|
||||||
|
}
|
||||||
@@ -0,0 +1,57 @@
|
|||||||
|
import { open } from "fs/promises";
|
||||||
|
import type { Client } from "tdl";
|
||||||
|
import { childLogger } from "../util/logger.js";
|
||||||
|
import { withFloodWait } from "../util/retry.js";
|
||||||
|
|
||||||
|
const log = childLogger("range-download");
|
||||||
|
const RANGE_TIMEOUT_MS = 120_000;
|
||||||
|
|
||||||
|
// NOTE (from Task 4 spike): TDLib writes the requested region into
|
||||||
|
// file.local.path at its ABSOLUTE file offset; file.local.downloaded_prefix_size
|
||||||
|
// counts contiguous bytes from download_offset. We request a 1KB-aligned offset
|
||||||
|
// so downloaded_prefix_size covers our whole [offset, offset+limit) window.
|
||||||
|
// This absolute-offset assumption is PENDING LIVE VERIFICATION ON DEPLOY —
|
||||||
|
// the authenticated TDLib session could not be spiked in this environment.
|
||||||
|
export async function downloadFileRange(
|
||||||
|
client: Client,
|
||||||
|
fileId: string,
|
||||||
|
offset: number,
|
||||||
|
limit: number,
|
||||||
|
expectedSize: bigint,
|
||||||
|
): Promise<Buffer> {
|
||||||
|
const numericId = parseInt(fileId, 10);
|
||||||
|
const alignedOffset = Math.max(0, offset - (offset % 1024));
|
||||||
|
const alignedLimit = limit + (offset - alignedOffset);
|
||||||
|
|
||||||
|
const file = await withFloodWait(
|
||||||
|
() =>
|
||||||
|
new Promise<{ local: { path: string; download_offset: number; downloaded_prefix_size: number } }>(
|
||||||
|
(resolve, reject) => {
|
||||||
|
const timer = setTimeout(() => reject(new Error(`Range download timed out for ${fileId}`)), RANGE_TIMEOUT_MS);
|
||||||
|
client
|
||||||
|
.invoke({
|
||||||
|
_: "downloadFile",
|
||||||
|
file_id: numericId,
|
||||||
|
priority: 1,
|
||||||
|
offset: alignedOffset,
|
||||||
|
limit: alignedLimit,
|
||||||
|
synchronous: true,
|
||||||
|
} as never)
|
||||||
|
.then((f) => { clearTimeout(timer); resolve(f as never); })
|
||||||
|
.catch((e) => { clearTimeout(timer); reject(e); });
|
||||||
|
},
|
||||||
|
),
|
||||||
|
`downloadFileRange:${fileId}`,
|
||||||
|
);
|
||||||
|
|
||||||
|
const start = offset;
|
||||||
|
const fh = await open(file.local.path, "r");
|
||||||
|
try {
|
||||||
|
const buf = Buffer.alloc(limit);
|
||||||
|
const { bytesRead } = await fh.read(buf, 0, limit, start);
|
||||||
|
log.debug({ fileId, offset, limit, bytesRead }, "range read");
|
||||||
|
return bytesRead < limit ? buf.subarray(0, bytesRead) : buf;
|
||||||
|
} finally {
|
||||||
|
await fh.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
+88
-31
@@ -63,9 +63,10 @@ import { hashParts } from "./archive/hash.js";
|
|||||||
import { readZipCentralDirectory } from "./archive/zip-reader.js";
|
import { readZipCentralDirectory } from "./archive/zip-reader.js";
|
||||||
import { readRarContents } from "./archive/rar-reader.js";
|
import { readRarContents } from "./archive/rar-reader.js";
|
||||||
import { read7zContents } from "./archive/sevenz-reader.js";
|
import { read7zContents } from "./archive/sevenz-reader.js";
|
||||||
|
import { tryProvenanceBackfill } from "./provenance-backfill.js";
|
||||||
import { byteLevelSplit, concatenateFiles } from "./archive/split.js";
|
import { byteLevelSplit, concatenateFiles } from "./archive/split.js";
|
||||||
import { uploadToChannel, UploadStallError } from "./upload/channel.js";
|
import { uploadToChannel, UploadStallError } from "./upload/channel.js";
|
||||||
import { processAlbumGroups, processRuleBasedGroups, processTimeWindowGroups, processPatternGroups, processCreatorGroups, processZipPathGroups, processReplyChainGroups, processCaptionGroups, detectGroupingConflicts, type IndexedPackageRef } from "./grouping.js";
|
import { processAlbumGroups, detectGroupingConflicts, type IndexedPackageRef } from "./grouping.js";
|
||||||
import { db } from "./db/client.js";
|
import { db } from "./db/client.js";
|
||||||
import type { TelegramAccount, TelegramChannel } from "@prisma/client";
|
import type { TelegramAccount, TelegramChannel } from "@prisma/client";
|
||||||
import type { Client } from "tdl";
|
import type { Client } from "tdl";
|
||||||
@@ -314,6 +315,7 @@ interface PipelineContext {
|
|||||||
zipsFound: number;
|
zipsFound: number;
|
||||||
zipsDuplicate: number;
|
zipsDuplicate: number;
|
||||||
zipsIngested: number;
|
zipsIngested: number;
|
||||||
|
zipsBackfilled: number;
|
||||||
};
|
};
|
||||||
/** Creator from forum topic name (null for non-forum). */
|
/** Creator from forum topic name (null for non-forum). */
|
||||||
topicCreator: string | null;
|
topicCreator: string | null;
|
||||||
@@ -422,6 +424,7 @@ export async function runWorkerForAccount(
|
|||||||
zipsFound: 0,
|
zipsFound: 0,
|
||||||
zipsDuplicate: 0,
|
zipsDuplicate: 0,
|
||||||
zipsIngested: 0,
|
zipsIngested: 0,
|
||||||
|
zipsBackfilled: 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
try {
|
try {
|
||||||
@@ -538,6 +541,8 @@ export async function runWorkerForAccount(
|
|||||||
currentActivity: `Enumerating topics in "${channelLabel}"`,
|
currentActivity: `Enumerating topics in "${channelLabel}"`,
|
||||||
currentStep: "scanning",
|
currentStep: "scanning",
|
||||||
currentChannel: channelLabel,
|
currentChannel: channelLabel,
|
||||||
|
currentTopicId: null,
|
||||||
|
currentAccountChannelMapId: null,
|
||||||
currentFile: null,
|
currentFile: null,
|
||||||
currentFileNum: null,
|
currentFileNum: null,
|
||||||
totalFiles: null,
|
totalFiles: null,
|
||||||
@@ -750,6 +755,8 @@ export async function runWorkerForAccount(
|
|||||||
currentActivity: `Scanning "${topicLabel}"${topicProgress}`,
|
currentActivity: `Scanning "${topicLabel}"${topicProgress}`,
|
||||||
currentStep: "scanning",
|
currentStep: "scanning",
|
||||||
currentChannel: channelLabel,
|
currentChannel: channelLabel,
|
||||||
|
currentTopicId: topic.topicId,
|
||||||
|
currentAccountChannelMapId: mapping.id,
|
||||||
currentFile: null,
|
currentFile: null,
|
||||||
currentFileNum: null,
|
currentFileNum: null,
|
||||||
totalFiles: null,
|
totalFiles: null,
|
||||||
@@ -830,7 +837,11 @@ export async function runWorkerForAccount(
|
|||||||
topic.name,
|
topic.name,
|
||||||
messageId
|
messageId
|
||||||
);
|
);
|
||||||
}
|
},
|
||||||
|
// shouldStop: re-read the live fetch flag before each archive
|
||||||
|
// set. A mid-run "disable topic" lets the current file finish,
|
||||||
|
// then skips the rest of this topic's archives.
|
||||||
|
async () => !(await isTopicFetchEnabled(mapping.id, topic.topicId))
|
||||||
);
|
);
|
||||||
// Sync client back in case it was recreated during upload stall recovery
|
// Sync client back in case it was recreated during upload stall recovery
|
||||||
client = pipelineCtx.client;
|
client = pipelineCtx.client;
|
||||||
@@ -922,6 +933,8 @@ export async function runWorkerForAccount(
|
|||||||
currentActivity: `Scanning "${channelLabel}" for new archives`,
|
currentActivity: `Scanning "${channelLabel}" for new archives`,
|
||||||
currentStep: "scanning",
|
currentStep: "scanning",
|
||||||
currentChannel: channelLabel,
|
currentChannel: channelLabel,
|
||||||
|
currentTopicId: null,
|
||||||
|
currentAccountChannelMapId: null,
|
||||||
currentFile: null,
|
currentFile: null,
|
||||||
currentFileNum: null,
|
currentFileNum: null,
|
||||||
totalFiles: null,
|
totalFiles: null,
|
||||||
@@ -1203,7 +1216,12 @@ async function processArchiveSets(
|
|||||||
* below any failed message ID in this scan). Used by the caller to
|
* below any failed message ID in this scan). Used by the caller to
|
||||||
* advance the channel/topic watermark incrementally — otherwise a long
|
* advance the channel/topic watermark incrementally — otherwise a long
|
||||||
* scan that gets killed by worker restart loses all progress. */
|
* scan that gets killed by worker restart loses all progress. */
|
||||||
onWatermarkAdvance?: (messageId: bigint) => Promise<void>
|
onWatermarkAdvance?: (messageId: bigint) => Promise<void>,
|
||||||
|
/** Optional cancellation check, polled before each archive set. When it
|
||||||
|
* resolves true, processing stops after the set currently in flight (that
|
||||||
|
* one completes; remaining sets in this scan are skipped). Used by the
|
||||||
|
* forum branch to honour a mid-run "disable topic". */
|
||||||
|
shouldStop?: () => Promise<boolean>
|
||||||
): Promise<{ maxProcessedId: bigint | null; minFailedId: bigint | null }> {
|
): Promise<{ maxProcessedId: bigint | null; minFailedId: bigint | null }> {
|
||||||
const { client, runId, channelTitle, channel, throttled, counters, accountLog } = ctx;
|
const { client, runId, channelTitle, channel, throttled, counters, accountLog } = ctx;
|
||||||
|
|
||||||
@@ -1281,6 +1299,16 @@ async function processArchiveSets(
|
|||||||
const indexedPackageRefs: IndexedPackageRef[] = [];
|
const indexedPackageRefs: IndexedPackageRef[] = [];
|
||||||
|
|
||||||
for (let setIdx = 0; setIdx < archiveSets.length; setIdx++) {
|
for (let setIdx = 0; setIdx < archiveSets.length; setIdx++) {
|
||||||
|
// Cooperative cancellation: if the caller signals stop (e.g. the topic was
|
||||||
|
// disabled mid-run), skip the remaining archive sets in this scan. The set
|
||||||
|
// processed in the previous iteration has already completed.
|
||||||
|
if (shouldStop && (await shouldStop())) {
|
||||||
|
accountLog.info(
|
||||||
|
{ channel: channelTitle, processed: setIdx, total: archiveSets.length },
|
||||||
|
"Stop signal received (topic disabled) — skipping remaining archive sets in this scan"
|
||||||
|
);
|
||||||
|
break;
|
||||||
|
}
|
||||||
try {
|
try {
|
||||||
const packageId = await processOneArchiveSet(
|
const packageId = await processOneArchiveSet(
|
||||||
ctx,
|
ctx,
|
||||||
@@ -1454,34 +1482,13 @@ async function processArchiveSets(
|
|||||||
scanResult.photos
|
scanResult.photos
|
||||||
);
|
);
|
||||||
|
|
||||||
// Auto-grouping passes (gated by per-channel flag)
|
// Heuristic auto-grouping passes (rule/time/pattern/creator/zip-path/
|
||||||
const channelRecord = await db.telegramChannel.findUnique({
|
// reply-chain/caption) were removed: the STL view is now a flat list
|
||||||
where: { id: channel.id },
|
// organized by the creator filter, so automatically inventing groups at
|
||||||
select: { autoGroupEnabled: true },
|
// ingestion is no longer wanted. Album grouping above is kept because it
|
||||||
});
|
// reflects real upload structure (files posted together as one Telegram
|
||||||
|
// album), not a heuristic guess. Existing groups and the manual grouping
|
||||||
if (channelRecord?.autoGroupEnabled !== false) {
|
// actions in the UI are unaffected.
|
||||||
// Learned rule-based grouping (from manual overrides)
|
|
||||||
await processRuleBasedGroups(channel.id, indexedPackageRefs);
|
|
||||||
|
|
||||||
// Time-window grouping for remaining ungrouped packages
|
|
||||||
await processTimeWindowGroups(channel.id, indexedPackageRefs);
|
|
||||||
|
|
||||||
// Pattern-based grouping (date patterns, project slugs)
|
|
||||||
await processPatternGroups(channel.id, indexedPackageRefs);
|
|
||||||
|
|
||||||
// Creator-based grouping (3+ files from same creator)
|
|
||||||
await processCreatorGroups(channel.id, indexedPackageRefs);
|
|
||||||
|
|
||||||
// ZIP path prefix grouping (shared root folder inside archives)
|
|
||||||
await processZipPathGroups(channel.id, indexedPackageRefs);
|
|
||||||
|
|
||||||
// Reply chain grouping (messages replying to same root)
|
|
||||||
await processReplyChainGroups(channel.id, indexedPackageRefs);
|
|
||||||
|
|
||||||
// Caption fuzzy match grouping
|
|
||||||
await processCaptionGroups(channel.id, indexedPackageRefs);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Check for potential grouping conflicts
|
// Check for potential grouping conflicts
|
||||||
await detectGroupingConflicts(channel.id, indexedPackageRefs);
|
await detectGroupingConflicts(channel.id, indexedPackageRefs);
|
||||||
@@ -1641,6 +1648,56 @@ async function processOneArchiveSet(
|
|||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ── Cross-channel provenance backfill ──
|
||||||
|
// The same-channel checks above missed. Before downloading, see if this
|
||||||
|
// archive is the true origin of a placeholder-source package (manual upload
|
||||||
|
// / rebuild record whose sourceChannelId == destChannelId). If so, backfill
|
||||||
|
// its real provenance and skip the download entirely.
|
||||||
|
const archType = archiveSet.type === "7Z" ? "SEVEN_Z" : archiveSet.type;
|
||||||
|
if (destChannelId && (archType === "ZIP" || archType === "RAR" || archType === "SEVEN_Z")) {
|
||||||
|
try {
|
||||||
|
const derivedCreator =
|
||||||
|
topicCreator && topicCreator !== "General"
|
||||||
|
? topicCreator
|
||||||
|
: (extractCreatorFromFileName(archiveName) ?? topicCreator ?? null);
|
||||||
|
const preview = previewMatches.get(archiveSet.baseName);
|
||||||
|
const result = await tryProvenanceBackfill({
|
||||||
|
client,
|
||||||
|
destChannelId,
|
||||||
|
scannedSourceChannelId: channel.id,
|
||||||
|
fileName: archiveName,
|
||||||
|
fileSize: totalArchiveSize,
|
||||||
|
archiveType: archType,
|
||||||
|
sourceMessageId: archiveSet.parts[0].id,
|
||||||
|
sourceTopicId,
|
||||||
|
sourceCaption: archiveSet.parts[0].caption ?? null,
|
||||||
|
remoteUniqueId: archiveSet.parts[0].remoteUniqueId ?? null,
|
||||||
|
creator: derivedCreator,
|
||||||
|
scannedParts: archiveSet.parts.map((p) => ({ fileId: p.fileId, fileSize: p.fileSize, fileName: p.fileName })),
|
||||||
|
previewData: null,
|
||||||
|
previewMsgId: preview?.id ?? null,
|
||||||
|
});
|
||||||
|
if (result.backfilled) {
|
||||||
|
counters.zipsBackfilled++;
|
||||||
|
accountLog.info(
|
||||||
|
{ fileName: archiveName, sourceMessageId: Number(archiveSet.parts[0].id), confidence: result.confidence },
|
||||||
|
"Backfilled provenance for placeholder package — skipping download",
|
||||||
|
);
|
||||||
|
await updateRunActivity(runId, {
|
||||||
|
currentActivity: `Backfilled provenance for ${archiveName}`,
|
||||||
|
currentStep: "backfilling",
|
||||||
|
currentFile: archiveName,
|
||||||
|
currentFileNum: setIdx + 1,
|
||||||
|
totalFiles: totalSets,
|
||||||
|
zipsBackfilled: counters.zipsBackfilled,
|
||||||
|
});
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
} catch (err) {
|
||||||
|
accountLog.warn({ err, fileName: archiveName }, "Provenance backfill attempt failed (non-fatal), continuing to normal ingestion");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ── Size guard: skip archives that exceed WORKER_MAX_ZIP_SIZE_MB ──
|
// ── Size guard: skip archives that exceed WORKER_MAX_ZIP_SIZE_MB ──
|
||||||
const maxSizeBytes = BigInt(config.maxZipSizeMB) * 1024n * 1024n;
|
const maxSizeBytes = BigInt(config.maxZipSizeMB) * 1024n * 1024n;
|
||||||
if (totalArchiveSize > maxSizeBytes) {
|
if (totalArchiveSize > maxSizeBytes) {
|
||||||
|
|||||||
@@ -0,0 +1,8 @@
|
|||||||
|
import { defineConfig } from "vitest/config";
|
||||||
|
|
||||||
|
export default defineConfig({
|
||||||
|
test: {
|
||||||
|
environment: "node",
|
||||||
|
include: ["src/**/*.test.ts"],
|
||||||
|
},
|
||||||
|
});
|
||||||
Reference in New Issue
Block a user