diff --git a/.gitignore b/.gitignore index 42a7b6a..c890cdd 100644 --- a/.gitignore +++ b/.gitignore @@ -1,12 +1,12 @@ -# tdl chat exports and the narrowed subsets built from them. These hold real -# message ids, file names and text from an account, and are specific to one run. -# The repo ships no JSON of its own, so the whole extension is excluded. +# Leftovers from the retired shell pipeline: chat exports and the narrowed +# subsets built from them. These hold real message ids, file names and text from +# an account. The Go binary reads the chat live and writes none of them, but the +# files may still be sitting in a working copy. The repo ships no JSON of its +# own, so the whole extension is excluded. *.json - -# Ids still to fetch, written by verify-export.sh. missing-ids.txt -# Staging holds media in flight between tdl and the remote. +# Staging holds media in flight between the download and upload legs. staging/ # Run logs carry chat names and progress output. diff --git a/export-until-complete.sh b/export-until-complete.sh deleted file mode 100755 index 3095f32..0000000 --- a/export-until-complete.sh +++ /dev/null @@ -1,113 +0,0 @@ -#!/usr/bin/env bash -# -# Drive run.sh repeatedly until every media file in a chat is on the remote. -# -# Each pass verifies what is already there, narrows the export JSON to just the -# ids still needed, and runs the pipeline on that subset. It stops when the -# verifier reports complete, when a pass makes no progress (the remaining media -# is genuinely unavailable on Telegram's side), or when the remote runs low on -# space. -# -# Usage: ./export-until-complete.sh -r REMOTE:PATH -c CHAT [options] -# -r REMOTE:PATH rclone destination (required) -# -c CHAT chat id/username, used for the initial metadata export -# -f FILE export JSON (default export-.json) -# -d DIR staging directory (default ./staging) -# -i SECONDS rclone sweep interval (default 120) -# -p N maximum passes (default 30) -# -m SIZE cap the staging directory at SIZE, e.g. 40G (default: no cap) -# -q GIB stop if remote free space falls below this (default 5) - -set -euo pipefail - -remote='' chat='' export_file='' staging='./staging' interval=120 max_staging='' -max_passes=30 min_free_gib=5 - -while getopts ':r:c:f:d:i:p:q:m:h' o; do case $o in - r) remote=$OPTARG ;; c) chat=$OPTARG ;; f) export_file=$OPTARG ;; - d) staging=$OPTARG ;; i) interval=$OPTARG ;; p) max_passes=$OPTARG ;; - q) min_free_gib=$OPTARG ;; m) max_staging=$OPTARG ;; - h) sed -n '2,19p' "$0"; exit 0 ;; - *) echo "usage: $0 -r REMOTE:PATH -c CHAT [-f FILE] [-i SECS] [-p N] [-q GIB] [-m SIZE]" >&2; exit 2 ;; -esac; done - -log() { printf '\n=== %s [driver] %s ===\n' "$(date '+%Y-%m-%d %H:%M:%S')" "$*"; } - -[[ -n $remote ]] || { echo "-r REMOTE:PATH is required" >&2; exit 2; } -[[ -n $chat || -n $export_file ]] || { echo "-c CHAT or -f FILE is required" >&2; exit 2; } -[[ -n $export_file ]] || export_file="export-${chat}.json" - -# Stop the whole loop on Ctrl-C rather than rolling into the next pass. run.sh -# installs its own handlers, so a pass shuts down cleanly before we exit. -interrupted=0 -trap 'interrupted=1' INT TERM - -# tdl's progress bar is ANSI redraws: great on a terminal, unreadable in a log. -# Show it when stdout is a TTY, suppress it when output is redirected. -tdl_quiet=() -[[ -t 1 ]] || tdl_quiet=(--disable-progress-ps) - -# The metadata export must exist before the first verify has anything to compare. -if [[ ! -f $export_file ]]; then - [[ -n $chat ]] || { echo "$export_file missing and no -c CHAT to create it" >&2; exit 2; } - log "exporting $chat metadata to $export_file" - tdl chat export -c "$chat" --all --with-content -o "$export_file" -fi - -free_gib() { - rclone about "${remote%%:*}:" --json 2>/dev/null \ - | python3 -c 'import json,sys; print(int(json.load(sys.stdin).get("free",0))//2**30)' 2>/dev/null \ - || echo 999999 # backends without quota reporting must not block the run -} - -prev_todo=-1 -for ((pass = 1; pass <= max_passes; pass++)); do - ((interrupted)) && { log 'interrupted; stopping'; exit 130; } - - log "pass $pass/$max_passes: verifying" - if ./verify-export.sh -f "$export_file" -r "$remote" -d "$staging"; then - log "complete after $((pass - 1)) download pass(es)" - exit 0 - fi - - todo=$(wc -l < missing-ids.txt) - if ((todo == prev_todo)); then - log "no progress in the last pass; $todo file(s) look permanently unavailable" - log 'ids left in missing-ids.txt' - exit 1 - fi - prev_todo=$todo - - free=$(free_gib) - if ((free < min_free_gib)); then - log "remote has only ${free} GiB free (limit ${min_free_gib}); stopping before it fills" - exit 3 - fi - log "$todo file(s) to fetch; ${free} GiB free on remote" - - # Narrow the full export to the outstanding ids, preserving its top-level shape - # so tdl reads it exactly like the original. - python3 - "$export_file" <<'PY' -import json, sys -d = json.load(open(sys.argv[1])) -want = {int(l) for l in open('missing-ids.txt') if l.strip()} -d['messages'] = [m for m in d['messages'] if m['id'] in want] -json.dump(d, open('gap.json', 'w')) -print(f"gap.json: {len(d['messages'])} messages") -PY - - rc=0 - ./run.sh -r "$remote" -f gap.json -d "$staging" -i "$interval" \ - ${max_staging:+-m "$max_staging"} \ - -- --group=false -t 4 -l 2 ${tdl_quiet[@]+"${tdl_quiet[@]}"} || rc=$? - log "pass $pass finished (run.sh exit $rc)" - - case $rc in - 0|1) ;; # done or partial: verify decides - 3) log 'rclone failure (remote full or unreachable); stopping'; exit 3 ;; - 130|143) log 'run interrupted; stopping'; exit 130 ;; - esac -done - -log "hit the $max_passes-pass limit; run again to continue" -exit 1 diff --git a/run.sh b/run.sh deleted file mode 100755 index b552068..0000000 --- a/run.sh +++ /dev/null @@ -1,333 +0,0 @@ -#!/usr/bin/env bash -# -# Rolling pipeline: tdl downloads Telegram media into a small staging directory -# while rclone concurrently moves finished files to any rclone remote (S3, -# Google Drive, SFTP, WebDAV, B2, ...). Local disk only ever holds the files in -# flight plus one sync interval of throughput, so a chat larger than the local -# disk can still be exported. -# -# See README.md for the background. -# -# Exit codes: 0 ok, 2 usage error, 3 rclone failure, 130/143 interrupted, -# anything else is tdl's own exit code. - -set -euo pipefail - -readonly PROG=${0##*/} - -# tdl downloads to '.tmp' and renames only on completion, so an unfinished -# file is always identifiable by extension. Never move one: a download stalled -# by a flood wait stops touching its .tmp, which then ages past --min-age and -# would be uploaded half-written, destroying tdl's resume point for that file. -readonly TEMP_GLOB='*.tmp' - -# Defaults -export_file='' # decided after parsing: per-chat when -c is given -chat='' -staging='./staging' -remote='' -interval=60 -min_age='45s' -max_staging='' # empty: staging grows as fast as tdl fills it -max_sync_failures=5 -cap_check_interval=10 # seconds between -m checks, independent of -i - -usage() { - cat <.json with -c, - otherwise export.json) - -c CHAT chat to export when FILE does not exist. Accepts a numeric - id as printed by 'tdl chat ls', a username with or without - '@', or a t.me/tg:// link. A Bot API '-100...' id is - converted to the plain id tdl expects. - -d DIR staging directory (default: $staging) - -i SECONDS seconds between rclone sweeps (default: $interval) - -a AGE rclone --min-age, a second guard against moving files still - being written (default: $min_age) - -m SIZE cap the staging directory at SIZE (K/M/G/T, binary; e.g. - 40G). When staging reaches it, tdl is suspended until - rclone has drained the finished files. Unset means no cap. - -h this help - -Everything after -- is appended to the 'tdl dl' command, e.g. - $PROG -r gdrive:telegram/media -- -t 4 -l 1 -USAGE -} - -log() { printf '%s [%s] %s\n' "$(date '+%Y-%m-%d %H:%M:%S')" "$PROG" "$*" >&2; } -die() { log "error: $*"; exit 2; } # usage or precondition -fail() { log "error: $*"; exit 3; } # rclone / pipeline failure - -while getopts ':f:c:d:r:i:a:m:h' opt; do - case $opt in - f) export_file=$OPTARG ;; - c) chat=$OPTARG ;; - d) staging=$OPTARG ;; - r) remote=$OPTARG ;; - i) interval=$OPTARG ;; - a) min_age=$OPTARG ;; - m) max_staging=$OPTARG ;; - h) usage; exit 0 ;; - :) die "option -$OPTARG requires an argument" ;; - ?) die "unknown option -$OPTARG (try -h)" ;; - esac -done -shift $((OPTIND - 1)) -tdl_extra=("$@") - -[[ -n $remote ]] || { usage >&2; die "-r REMOTE:PATH is required"; } -[[ $remote == *:* ]] || die "remote '$remote' is not in rclone REMOTE:PATH form" -[[ $interval =~ ^[0-9]+$ && $interval -gt 0 ]] || die "-i must be a positive integer" -[[ $min_age =~ ^[0-9]+(\.[0-9]+)?(ms|s|m|h|d|w|M|y)?$ ]] \ - || die "-a must be an rclone duration, e.g. 2m" - -# Sizes are compared in KiB because that is the unit 'du -sk' reports, which is -# also the unit that matters here: allocated blocks, not apparent length. -max_staging_kib='' -if [[ -n $max_staging ]]; then - [[ $max_staging =~ ^[0-9]+[KkMmGgTt]?$ ]] \ - || die "-m must be a size with an optional K/M/G/T suffix, e.g. 40G" - num=${max_staging%[KkMmGgTt]} - case ${max_staging#"$num"} in - ''|K|k) max_staging_kib=$num ;; - M|m) max_staging_kib=$((num * 1024)) ;; - G|g) max_staging_kib=$((num * 1024 * 1024)) ;; - T|t) max_staging_kib=$((num * 1024 * 1024 * 1024)) ;; - esac - ((max_staging_kib > 0)) || die "-m must be greater than zero" -fi - -# tdl resolves a numeric argument as an MTProto id and anything else through -# gotd's resolver, which handles '@name', 'name' and t.me/tg:// links. Two forms -# still need help: Bot API ids carry a '-100' prefix that MTProto does not use, -# and a message link is not a chat. -if [[ -n $chat ]]; then - read -r chat <<<"$chat" # trim stray whitespace - case $chat in - '') die "-c requires a chat" ;; - *t.me/c/*|*t.me/*/[0-9]*) - die "-c takes a chat, not a message link ($chat) — pass the chat's username or id" ;; - -100[0-9]*) - log "converting Bot API id $chat to MTProto id ${chat#-100}" - chat=${chat#-100} ;; - esac -fi - -# Keep each chat's export in its own file, so switching -c never silently -# downloads the previous chat again from a stale export.json. -if [[ -z $export_file ]]; then - if [[ -n $chat ]]; then - slug=${chat#@} - slug=${slug##*/} - slug=$(printf '%s' "$slug" | tr -c 'A-Za-z0-9._-' '_') - export_file="export-$slug.json" - else - export_file='export.json' - fi -fi - -# pikpak-style backends finish an upload as a server-side async task, and -# rclone abandons a still-pending one once --low-level-retries polls run out, -# failing a transfer that would have succeeded. Fewer parallel transfers keep -# that queue short and more retries wait it out. These are rclone's own -# environment variables, so whatever the caller already exported wins. -: "${RCLONE_TRANSFERS:=2}" -: "${RCLONE_LOW_LEVEL_RETRIES:=20}" -export RCLONE_TRANSFERS RCLONE_LOW_LEVEL_RETRIES - -for tool in tdl rclone; do - command -v "$tool" >/dev/null || die "$tool is not installed or not on PATH" -done - -# A named remote must exist in the config; a leading ':' means an on-the-fly -# connection string, which has no config entry to check. Listing the remote's -# root is not portable (some backends refuse it), so reachability and -# credentials are proven by creating the destination, which rclone would create -# on the first move anyway. -if [[ $remote != :* ]]; then - # Read the list into a variable first: piping it into 'grep -q' lets grep exit - # on the first match and kill rclone with SIGPIPE, which pipefail then reports - # as a failed pipeline -- rejecting a remote that is in fact configured. - remotes=$(rclone listremotes 2>/dev/null || true) - grep -qx -- "${remote%%:*}:" <<<"$remotes" \ - || die "rclone remote '${remote%%:*}:' is not configured — see 'rclone listremotes'" -fi -rclone mkdir "$remote" >/dev/null 2>&1 \ - || die "cannot reach '$remote' — check credentials and connectivity" - -# The export JSON only lists messages; it is cheap to keep and required for both -# legs to stay resumable, so never regenerate it when it already exists. -if [[ ! -f $export_file ]]; then - [[ -n $chat ]] || die "$export_file not found; pass -c CHAT to export it first" - log "exporting $chat metadata to $export_file" - tdl chat export -c "$chat" --all --with-content -o "$export_file" -else - log "using existing $export_file (delete it to re-export)" -fi - -mkdir -p "$staging" - -tdl_pid='' -tdl_rc=0 -sweep_ok=0 - -# The sweeps that run once at the end have a whole staging directory to move -# and no interleaved tdl output, so they report progress instead of going quiet -# for several minutes. rclone's redrawn bar is right on a terminal but turns a -# redirected log into control characters, so a log gets periodic one-line -# stats. Those are logged at INFO, which '-v' would enable at the cost of a -# line per file, hence raising the stats to NOTICE rather than the whole log. -sweep_progress=(--progress) -[[ -t 1 ]] || sweep_progress=(--stats 30s --stats-one-line --stats-log-level NOTICE) - -# Move whatever is finished. Partial .tmp files are always excluded. $1 is an -# optional --min-age guard; $2 enables --delete-empty-src-dirs, which is safe -# only once tdl has stopped — rclone removing a directory between tdl's -# MkdirAll and Create makes tdl fail, and it can take the staging root too, -# hence the mkdir afterwards. $3 turns on the progress reporting above. -sweep() { - local rc=0 - local args=(--exclude "$TEMP_GLOB") - if [[ -n ${1:-} ]]; then args+=(--min-age "$1"); fi - if ((${2:-0})); then args+=(--delete-empty-src-dirs); fi - if ((${3:-0})); then args+=(${sweep_progress[@]+"${sweep_progress[@]}"}); fi - rclone move "$staging" "$remote" "${args[@]}" || rc=$? - mkdir -p "$staging" - return $rc -} - -staging_kib() { du -sk "$staging" 2>/dev/null | cut -f1; } - -over_cap() { - local used - used=$(staging_kib) - [[ -n $used ]] && ((used >= max_staging_kib)) -} - -# Enforce -m. tdl renames a file only once it is complete, so staging holds -# finished files plus the in-flight '*.tmp' ones, and only the former can be -# drained. Suspending tdl stops it adding more while rclone empties the -# directory; SIGSTOP is safe because tdl reconnects on resume and --continue -# picks its .tmp files back up. The cap must therefore stay well above what the -# concurrent downloads hold, or draining could never clear it. -drain_to_cap() { - local used rc=0 - used=$(staging_kib) - log "staging at $((used / 1024)) MiB, at or over the $((max_staging_kib / 1024)) MiB cap; suspending tdl to drain" - - kill -STOP "$tdl_pid" 2>/dev/null || true - # Sweep until staging is back under the cap, since one rclone move need not - # get there: a slow remote or a per-file error can leave finished files - # behind. Stop as soon as a sweep frees nothing, which means all that is - # left is in-flight .tmp files that no sweep can ever move. - local before - while :; do - before=$used - # No --min-age: tdl is frozen, so every non-.tmp file is finished by - # construction and waiting out the guard would only prolong the pause. - sweep '' || { rc=$?; break; } - used=$(staging_kib) - [[ -n $used ]] || break - ((used >= max_staging_kib && used < before)) || break - done - kill -CONT "$tdl_pid" 2>/dev/null || true - - if ((rc == 0)) && [[ -n $used ]] && ((used >= max_staging_kib)); then - log "warning: staging is still $((used / 1024)) MiB after draining — the cap is below what the in-flight downloads hold; raise -m or lower tdl's -l/-t" - else - log "resumed tdl, staging now $((${used:-0} / 1024)) MiB" - fi - return $rc -} - -cleanup() { - if [[ -n $tdl_pid ]] && kill -0 "$tdl_pid" 2>/dev/null; then - log "stopping tdl (pid $tdl_pid)" - # A suspended process never sees SIGTERM, so let it run first. - kill -CONT "$tdl_pid" 2>/dev/null || true - kill -TERM "$tdl_pid" 2>/dev/null || true - for _ in 1 2 3 4 5 6 7 8 9 10; do - kill -0 "$tdl_pid" 2>/dev/null || break - sleep 1 - done - kill -KILL "$tdl_pid" 2>/dev/null || true - fi - if ((sweep_ok)); then - log 'sweeping completed files before exit' - sweep "$min_age" 0 1 || log 'warning: final safety sweep failed; staging kept' - fi -} -trap cleanup EXIT -trap 'log "interrupted (SIGINT)"; exit 130' INT -trap 'log "terminated (SIGTERM)"; exit 143' TERM - -log "downloading into $staging, moving to $remote every ${interval}s${max_staging_kib:+, capped at $((max_staging_kib / 1024)) MiB}" -tdl dl -f "$export_file" -d "$staging" \ - --takeout --group --skip-same --continue ${tdl_extra[@]+"${tdl_extra[@]}"} & -tdl_pid=$! -sweep_ok=1 - -failures=0 -waited=0 -while kill -0 "$tdl_pid" 2>/dev/null; do - # A long sweep interval must not let staging blow past the cap in between, so - # sleep in slices and check the cap on each one. Every sleep is a job, so - # signals are handled without waiting the slice out. - slice=$((interval - waited)) - if [[ -n $max_staging_kib ]] && ((slice > cap_check_interval)); then - slice=$cap_check_interval - fi - sleep "$slice" & - wait $! 2>/dev/null || true - waited=$((waited + slice)) - - # Only a sweep that actually ran says anything about rclone's health, so the - # failure streak is judged on those alone and a quiet cap check never - # clears it. - rc=0 - swept=0 - if ((waited >= interval)); then - waited=0 - swept=1 - sweep "$min_age" || rc=$? - fi - if ((rc == 0)) && [[ -n $max_staging_kib ]] && kill -0 "$tdl_pid" 2>/dev/null; then - if over_cap; then - swept=1 - drain_to_cap || rc=$? - fi - fi - - if ((swept)); then - if ((rc == 0)); then - failures=0 - else - failures=$((failures + 1)) - log "warning: rclone sweep failed ($failures/$max_sync_failures)" - if ((failures >= max_sync_failures)); then - sweep_ok=0 - fail "rclone failed $failures times in a row; stopping before staging fills the disk" - fi - fi - fi -done - -wait "$tdl_pid" || tdl_rc=$? -tdl_pid='' -sweep_ok=0 - -if ((tdl_rc != 0)); then - log "tdl exited $tdl_rc; staging kept at $staging — re-run to resume" - sweep "$min_age" 0 1 || log 'warning: sweep after failure did not complete' - exit "$tdl_rc" -fi - -log 'tdl finished; final sweep' -sweep '' 1 1 || fail "final sweep failed; files remain in $staging" -log "done — everything moved to $remote" diff --git a/verify-export.sh b/verify-export.sh deleted file mode 100755 index 064d123..0000000 --- a/verify-export.sh +++ /dev/null @@ -1,114 +0,0 @@ -#!/usr/bin/env bash -# -# Verify a tdl/rclone export is complete and every file is intact. -# -# Rebuilds the exact filename tdl produces for each media message in the export -# JSON ({DialogID}_{MessageID}_{FileName}) and checks it exists on the remote or -# in staging. The export is Telegram's own record of the name, so the match is -# exact: a file stored under any other name is not the file the export asked -# for and counts as absent, however close the name looks. Zero-byte files count -# as missing too: rclone overwrites a size-mismatched destination, so re-running -# repairs them. Files under 1 KiB are reported for inspection but trusted, since -# some real media is genuinely that small. -# -# A remote file whose id matches but whose name does not is a stale copy from an -# earlier download; it is listed separately so it can be deleted, because the -# re-download lands beside it rather than replacing it. -# -# Writes every id needing another attempt to missing-ids.txt. -# Exit 0 = complete, 1 = incomplete, 2 = usage error. -# -# Usage: ./verify-export.sh -f export-.json -r REMOTE:PATH [-d STAGING] - -set -euo pipefail - -export_file='' remote='' staging='./staging' -while getopts ':f:r:d:h' o; do case $o in - f) export_file=$OPTARG ;; r) remote=$OPTARG ;; d) staging=$OPTARG ;; - h) sed -n '2,16p' "$0"; exit 0 ;; - *) echo "usage: $0 -f FILE -r REMOTE:PATH [-d STAGING]" >&2; exit 2 ;; -esac; done - -[[ -f $export_file ]] || { echo "no such export file: $export_file" >&2; exit 2; } -[[ -n $remote ]] || { echo "-r REMOTE:PATH is required" >&2; exit 2; } - -listing=$(mktemp); trap 'rm -f "$listing"' EXIT -# lsl gives sizes as well as names, so truncated uploads are detectable. -rclone lsl "$remote" > "$listing" - -python3 - "$export_file" "$listing" "$staging" <<'PY' -import json, os, re, sys -export_file, listing, staging = sys.argv[1], sys.argv[2], sys.argv[3] - -sizes = {} -for line in open(listing): - m = re.match(r'^\s*(\d+)\s+\S+\s+\S+\s+(.*)$', line.rstrip('\n')) - if m: - sizes[os.path.basename(m.group(2))] = int(m.group(1)) -if os.path.isdir(staging): - for f in os.listdir(staging): - if not f.endswith('.tmp'): - sizes.setdefault(f, os.path.getsize(os.path.join(staging, f))) - -msgs = json.load(open(export_file))['messages'] -# Text-only messages carry an empty "file" and are not download targets. -media = [m for m in msgs if m.get('file')] - -dialog = str(json.load(open(export_file)).get('id', '')).lstrip('-') -if not dialog.isdigit(): - ids = {n.split('_')[0] for n in sizes if '_' in n} - dialog = ids.pop() if len(ids) == 1 else '' -if not dialog: - sys.exit('cannot determine dialog id') - -# Names already stored for each message id, used only to tell an absent file -# apart from one sitting there under the wrong name. -stored = {} -for name in sizes: - parts = name.split('_', 2) - if len(parts) == 3 and parts[0] == dialog and parts[1].isdigit(): - stored.setdefault(int(parts[1]), []).append(name) - -absent, empty, tiny, misnamed = [], [], [], [] -for m in media: - name = f"{dialog}_{m['id']}_{m['file']}" - if name not in sizes: - absent.append(m['id']) - for other in stored.get(m['id'], []): - misnamed.append((m['id'], name, other, sizes[other])) - elif sizes[name] == 0: - empty.append(m['id']) - elif sizes[name] < 1024: - tiny.append((m['id'], sizes[name], name)) - -todo = sorted(absent + empty) -print(f"messages in export : {len(msgs)}") -print(f" text-only (skip) : {len(msgs) - len(media)}") -print(f" media expected : {len(media)}") -print(f"present and intact : {len(media) - len(todo)}") -print(f" absent : {len(absent)}") -print(f" zero-byte : {len(empty)}") -if tiny: - print(f" under 1KiB (check, not retried): {len(tiny)}") - for i, s, n in tiny[:5]: - print(f" id {i} {s} B {n}") -if misnamed: - print(f"\nstored under a different name : {len(misnamed)}") - print(" counted as absent and fetched again; delete the stale copies so the") - print(" re-download does not leave two files for the same message:") - for i, want, got, size in misnamed: - print(f" id {i} {size} B") - print(f" export: {want}") - print(f" remote: {got}") - -if todo: - with open('missing-ids.txt', 'w') as fh: - fh.write('\n'.join(map(str, todo)) + '\n') - print(f"\nneeds another pass : {len(todo)} (ids {min(todo)}–{max(todo)})") - print("written to missing-ids.txt") - sys.exit(1) - -if os.path.exists('missing-ids.txt'): - os.remove('missing-ids.txt') -print("\nCOMPLETE: every media message is present and non-empty.") -PY