From b73b74efd5b12ddb913a3c01c0fa0a1951ea9944 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Sun, 6 Sep 2026 17:00:45 +0700 Subject: [PATCH 01/16] feat: add Go binary embedding tdl and rclone as libraries First slice of replacing the three-script pipeline with one process. The scripts coordinate tdl and rclone as separate programs, so everything expensive in them exists to work around the fact that neither can see the other's state. A single process does not need that machinery. This slice covers only the foundations: open the session tdl login already wrote, resolve an rclone destination, and report on both via a doctor command. Downloading, uploading and verification follow. The session store is shared with the tdl CLI rather than copied, so the two cannot run against one namespace at the same time; -n selects another. AppID and AppHash are read from the store rather than hardcoded, because a session is bound to the application that created it and tdl records which one it used. No middlewares are passed to tclient.New, which already prepends its own defaults; the DC pool gets them instead, since gotd applies a client's middlewares only to direct invocations and not to pooled connections. The rclone config is loaded up front because the lazy path calls os.Exit on a config it cannot read, which would bypass every defer and exit with the code this tool reserves for an incomplete run. Exit codes follow the shell pipeline: 0 ok, 1 incomplete, 2 usage, 3 remote failure, 130 SIGINT, 143 SIGTERM. Cancellation is checked explicitly after the Telegram client returns, because gotd reports an interrupted run as success and a driver would read that as a finished archive. --- .github/workflows/ci.yml | 40 ++ .gitignore | 4 + cmd/tgexport/doctor.go | 145 +++++++ cmd/tgexport/main.go | 153 ++++++++ cmd/tgexport/main_test.go | 100 +++++ go.mod | 253 +++++++++++++ go.sum | 737 ++++++++++++++++++++++++++++++++++++ internal/backends/all.go | 13 + internal/backends/slim.go | 11 + internal/remote/fs.go | 131 +++++++ internal/remote/fs_test.go | 117 ++++++ internal/tdlkv/bolt.go | 140 +++++++ internal/tdlkv/bolt_test.go | 179 +++++++++ internal/tgsource/client.go | 152 ++++++++ 14 files changed, 2175 insertions(+) create mode 100644 .github/workflows/ci.yml create mode 100644 cmd/tgexport/doctor.go create mode 100644 cmd/tgexport/main.go create mode 100644 cmd/tgexport/main_test.go create mode 100644 go.mod create mode 100644 go.sum create mode 100644 internal/backends/all.go create mode 100644 internal/backends/slim.go create mode 100644 internal/remote/fs.go create mode 100644 internal/remote/fs_test.go create mode 100644 internal/tdlkv/bolt.go create mode 100644 internal/tdlkv/bolt_test.go create mode 100644 internal/tgsource/client.go diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..6b866fa --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,40 @@ +name: ci + +on: + push: + branches: [main] + pull_request: + +jobs: + check: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-go@v5 + with: + go-version-file: go.mod + cache: true + + - name: gofmt + run: | + unformatted=$(gofmt -l ./cmd ./internal) + if [ -n "$unformatted" ]; then + echo "gofmt needed on:"; echo "$unformatted"; exit 1 + fi + + - name: vet + run: go vet ./... + + - name: test + run: go test ./... + + # Both tag combinations are compiled because they select different rclone + # backend sets. A dependency bump that breaks one can easily leave the + # other green, and rclone's library API is outside its compatibility + # promise, so this job is the tripwire for an unintended upgrade. + - name: compile (all backends) + run: go build ./... + + - name: compile (slim) + run: go build -tags slim ./... diff --git a/.gitignore b/.gitignore index f86285a..42a7b6a 100644 --- a/.gitignore +++ b/.gitignore @@ -14,3 +14,7 @@ staging/ # Session plans and reports. plans/ + +# Go build output +/tgexport +/tgexport-slim diff --git a/cmd/tgexport/doctor.go b/cmd/tgexport/doctor.go new file mode 100644 index 0000000..f39ceb4 --- /dev/null +++ b/cmd/tgexport/doctor.go @@ -0,0 +1,145 @@ +package main + +import ( + "context" + "errors" + "flag" + "fmt" + "runtime/debug" + "strings" + + "github.com/gotd/td/tg" + + "github.com/iyear/tdl/core/dcpool" + + "github.com/tiennm99dev/telegram-exporter/internal/remote" + "github.com/tiennm99dev/telegram-exporter/internal/tdlkv" + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" +) + +// doctorCmd reports whether the two halves this tool depends on are usable: an +// authorised tdl session, and a reachable destination remote. It is the cheapest +// way to separate "misconfigured" from "broken" before starting a long run. +// +// Both checks always run, so one broken half does not hide the state of the +// other, but their errors are returned rather than merely printed — a check +// tool that exits 0 on failure is worse than no check tool. +func doctorCmd(ctx context.Context, args []string) error { + fs := flag.NewFlagSet("doctor", flag.ContinueOnError) + var ( + remoteArg = fs.String("r", "", "rclone destination to check, e.g. pikpak:archive") + ns = fs.String("n", "default", "tdl session namespace") + dataDir = fs.String("storage", tdlkv.DefaultDir(), "tdl bolt storage directory") + ) + if err := fs.Parse(args); err != nil { + if errors.Is(err, flag.ErrHelp) { + return err // main maps this to a clean exit + } + return fmt.Errorf("%w: %v", errUsage, err) + } + + fmt.Printf("versions\n") + for _, dep := range []string{ + "github.com/iyear/tdl/core", + "github.com/rclone/rclone", + "github.com/gotd/td", + } { + fmt.Printf(" %-28s %s\n", dep, moduleVersion(dep)) + } + + var problems []error + + fmt.Printf("\ntelegram\n") + if err := checkTelegram(ctx, *dataDir, *ns); err != nil { + fmt.Printf(" session FAILED: %v\n", err) + problems = append(problems, fmt.Errorf("telegram: %w", err)) + } + + fmt.Printf("\nremote\n") + if *remoteArg == "" { + fmt.Printf(" skipped (pass -r REMOTE:PATH to check a destination)\n") + } else if err := checkRemote(ctx, *remoteArg); err != nil { + fmt.Printf(" %-9s FAILED: %v\n", *remoteArg, err) + problems = append(problems, fmt.Errorf("remote: %w", err)) + } + + return errors.Join(problems...) +} + +func checkTelegram(ctx context.Context, dataDir, ns string) error { + kv, err := tdlkv.Open(dataDir, ns) + if err != nil { + return err + } + defer func() { _ = kv.Close() }() + + fmt.Printf(" store %s (namespace %q)\n", dataDir, ns) + + sess, err := tgsource.New(ctx, tgsource.Options{KV: kv}) + if err != nil { + return err + } + + return sess.Run(ctx, func(ctx context.Context, _ dcpool.Pool) error { + self, err := sess.Client().Self(ctx) + if err != nil { + return fmt.Errorf("fetch self: %w", err) + } + fmt.Printf(" account %s (id %d)\n", describeUser(self), self.ID) + return nil + }) +} + +func checkRemote(ctx context.Context, dest string) error { + ctx, err := remote.Init(ctx, remote.DefaultTunables()) + if err != nil { + return err + } + + f, err := remote.Resolve(ctx, dest) + if err != nil { + return err + } + fmt.Printf(" resolved %s\n", f.String()) + + if free, ok := remote.FreeBytes(ctx, f); ok { + fmt.Printf(" free %.1f GiB\n", float64(free)/(1<<30)) + } else { + // Not a failure: several backends have no quota API, and the shell + // pipeline deliberately treated that as unlimited rather than blocking. + fmt.Printf(" free not reported by this backend (treated as unlimited)\n") + } + return nil +} + +func describeUser(u *tg.User) string { + parts := make([]string, 0, 3) + if u.FirstName != "" { + parts = append(parts, u.FirstName) + } + if u.LastName != "" { + parts = append(parts, u.LastName) + } + if u.Username != "" { + parts = append(parts, "@"+u.Username) + } + if len(parts) == 0 { + return "(unnamed)" + } + return strings.Join(parts, " ") +} + +// moduleVersion reports the version a dependency was built against, read from +// the binary itself so it cannot drift from what is actually linked in. +func moduleVersion(path string) string { + info, ok := debug.ReadBuildInfo() + if !ok { + return "unknown" + } + for _, d := range info.Deps { + if d.Path == path { + return d.Version + } + } + return "not linked" +} diff --git a/cmd/tgexport/main.go b/cmd/tgexport/main.go new file mode 100644 index 0000000..cc9738c --- /dev/null +++ b/cmd/tgexport/main.go @@ -0,0 +1,153 @@ +// Command tgexport archives Telegram chat media to an rclone remote. +// +// It replaces a three-script shell pipeline that ran `tdl dl` and `rclone move` +// as separate processes. Both are embedded here as libraries, so the program can +// see a download finish rather than inferring it from a filename suffix and a +// file's age. +package main + +import ( + "context" + "errors" + "flag" + "fmt" + "os" + "os/signal" + "syscall" + + // Registers the rclone storage backends this binary can talk to. Backend + // selection is a property of the binary, so the import lives here rather + // than in a library package where it would leak into every importer and + // make the `slim` build tag meaningless. + _ "github.com/tiennm99dev/telegram-exporter/internal/backends" +) + +// Exit codes, matching the shell pipeline so existing habits and any wrapper +// scripts keep working: run.sh used 0 ok, 2 usage, 3 rclone failure, 130 SIGINT, +// 143 SIGTERM, and export-until-complete.sh used 1 for "ran, still incomplete". +const ( + exitOK = 0 + exitIncomplete = 1 + exitUsage = 2 + exitRemoteError = 3 + exitSIGINT = 130 + exitSIGTERM = 143 +) + +// errUsage marks an error as the operator's mistake rather than a failure, +// selecting exit code 2. +var errUsage = errors.New("usage") + +// errIncomplete marks a run that finished cleanly but left work outstanding. +var errIncomplete = errors.New("incomplete") + +func main() { + os.Exit(run()) +} + +func run() int { + if len(os.Args) < 2 { + usage() + return exitUsage + } + if a := os.Args[1]; a == "-h" || a == "--help" || a == "help" { + usage() + return exitOK + } + + ctx, signalled := notifyContext() + + var err error + switch os.Args[1] { + case "doctor": + err = doctorCmd(ctx, os.Args[2:]) + default: + fmt.Fprintf(os.Stderr, "unknown command %q\n\n", os.Args[1]) + usage() + return exitUsage + } + + sig := signalled() + if sig != nil { + fmt.Fprintf(os.Stderr, "interrupted (%v)\n", sig) + } else if err != nil && !errors.Is(err, flag.ErrHelp) { + fmt.Fprintf(os.Stderr, "error: %v\n", err) + } + return exitCode(err, sig) +} + +// exitCode maps a command's outcome onto the shell pipeline's contract. +// +// A signal outranks whatever error the interruption produced on the way out: +// the operator stopped this, and the code has to say so rather than letting a +// driver read an abandoned run as finished. +func exitCode(err error, sig os.Signal) int { + if sig != nil { + if sig == syscall.SIGTERM { + return exitSIGTERM + } + return exitSIGINT + } + + switch { + case err == nil, errors.Is(err, flag.ErrHelp): + return exitOK + case errors.Is(err, context.Canceled): + return exitSIGINT + case errors.Is(err, errUsage): + return exitUsage + case errors.Is(err, errIncomplete): + return exitIncomplete + default: + return exitRemoteError + } +} + +// notifyContext cancels ctx on SIGINT or SIGTERM and reports which arrived. +// +// Unlike signal.NotifyContext it stops trapping after the first signal, so a +// second Ctrl-C kills the process outright. That matters when shutdown itself +// hangs — an rclone upload waiting on a slow pikpak commit, say — and the +// operator needs a way out that does not involve another terminal. +func notifyContext() (context.Context, func() os.Signal) { + ctx, cancel := context.WithCancel(context.Background()) + + ch := make(chan os.Signal, 1) + signal.Notify(ch, os.Interrupt, syscall.SIGTERM) + + var got os.Signal + done := make(chan struct{}) + finished := make(chan struct{}) + + go func() { + defer close(finished) + select { + case sig := <-ch: + got = sig + signal.Stop(ch) // next one gets the default disposition: die + cancel() + case <-done: + signal.Stop(ch) + cancel() + } + }() + + // Waiting on finished before reading got is what makes the read safe: the + // goroutine writes it and then closes the channel, so the happens-before + // edge is the close, not the return. + return ctx, func() os.Signal { + close(done) + <-finished + return got + } +} + +func usage() { + fmt.Fprint(os.Stderr, `Usage: tgexport [options] + +Commands: + doctor Check the Telegram session, the destination remote, and free space + +Run 'tgexport -h' for command options. +`) +} diff --git a/cmd/tgexport/main_test.go b/cmd/tgexport/main_test.go new file mode 100644 index 0000000..dfd0de3 --- /dev/null +++ b/cmd/tgexport/main_test.go @@ -0,0 +1,100 @@ +package main + +import ( + "context" + "errors" + "flag" + "fmt" + "os" + "syscall" + "testing" + "time" +) + +// The exit-code contract is what a driver script reads to decide whether the +// archive is finished. Every mapping is asserted here because the previous +// version of this code documented the contract in a comment and then returned +// nil on interruption, which a driver would have read as "complete". +func TestExitCode(t *testing.T) { + tests := []struct { + name string + err error + sig os.Signal + want int + }{ + {"success", nil, nil, exitOK}, + {"help is not a failure", flag.ErrHelp, nil, exitOK}, + {"wrapped help", fmt.Errorf("parse: %w", flag.ErrHelp), nil, exitOK}, + {"usage mistake", fmt.Errorf("%w: bad flag", errUsage), nil, exitUsage}, + {"run left work outstanding", fmt.Errorf("%w: 12 files", errIncomplete), nil, exitIncomplete}, + {"remote failure", errors.New("pikpak unreachable"), nil, exitRemoteError}, + {"cancelled without a signal", context.Canceled, nil, exitSIGINT}, + {"wrapped cancellation", fmt.Errorf("download: %w", context.Canceled), nil, exitSIGINT}, + {"SIGINT", nil, os.Interrupt, exitSIGINT}, + {"SIGTERM", nil, syscall.SIGTERM, exitSIGTERM}, + + // A signal outranks the error it produced. Without this, an interrupted + // run whose cleanup happened to fail would exit 3 and look like a remote + // problem instead of an operator stop. + {"signal outranks error", errors.New("upload aborted"), os.Interrupt, exitSIGINT}, + {"signal outranks success", nil, syscall.SIGTERM, exitSIGTERM}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + if got := exitCode(tt.err, tt.sig); got != tt.want { + t.Errorf("exitCode(%v, %v) = %d, want %d", tt.err, tt.sig, got, tt.want) + } + }) + } +} + +// Codes must not collide: a driver distinguishes them by value alone. +func TestExitCodesMatchShellPipeline(t *testing.T) { + // run.sh: 0 ok, 2 usage, 3 rclone failure, 130 SIGINT, 143 SIGTERM. + // export-until-complete.sh: 1 ran but still incomplete. + want := map[string]int{ + "ok": 0, "incomplete": 1, "usage": 2, "remote": 3, "sigint": 130, "sigterm": 143, + } + got := map[string]int{ + "ok": exitOK, "incomplete": exitIncomplete, "usage": exitUsage, + "remote": exitRemoteError, "sigint": exitSIGINT, "sigterm": exitSIGTERM, + } + for k, w := range want { + if got[k] != w { + t.Errorf("%s exit code = %d, want %d (shell pipeline contract)", k, got[k], w) + } + } +} + +func TestNotifyContextReportsSignal(t *testing.T) { + ctx, signalled := notifyContext() + + if err := syscall.Kill(os.Getpid(), syscall.SIGINT); err != nil { + t.Fatalf("send SIGINT: %v", err) + } + + select { + case <-ctx.Done(): + case <-time.After(5 * time.Second): + t.Fatal("context was not cancelled within 5s of SIGINT") + } + + if sig := signalled(); sig != syscall.SIGINT { + t.Errorf("signalled() = %v, want SIGINT", sig) + } +} + +// An uninterrupted command must not be reported as signalled, or every clean +// run would exit 130. +func TestNotifyContextReportsNoSignal(t *testing.T) { + ctx, signalled := notifyContext() + + if sig := signalled(); sig != nil { + t.Errorf("signalled() = %v on a clean run, want nil", sig) + } + // The accessor also releases the watcher, which cancels the context. + if ctx.Err() == nil { + t.Error("context should be cancelled once the watcher is released") + } +} diff --git a/go.mod b/go.mod new file mode 100644 index 0000000..c4cae8f --- /dev/null +++ b/go.mod @@ -0,0 +1,253 @@ +module github.com/tiennm99dev/telegram-exporter + +go 1.27.1 + +require ( + github.com/gotd/td v0.140.0 + github.com/iyear/tdl/core v0.20.4 + github.com/rclone/rclone v1.75.1 + go.etcd.io/bbolt v1.5.0 +) + +require ( + cloud.google.com/go/auth v0.20.0 // indirect + cloud.google.com/go/auth/oauth2adapt v0.2.8 // indirect + cloud.google.com/go/compute/metadata v0.9.0 // indirect + github.com/Azure/azure-sdk-for-go/sdk/azcore v1.22.0 // indirect + github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.14.0 // indirect + github.com/Azure/azure-sdk-for-go/sdk/internal v1.12.0 // indirect + github.com/Azure/azure-sdk-for-go/sdk/storage/azblob v1.8.0 // indirect + github.com/Azure/azure-sdk-for-go/sdk/storage/azfile v1.7.0 // indirect + github.com/Azure/go-ntlmssp v0.1.1 // indirect + github.com/AzureAD/microsoft-authentication-library-for-go v1.7.2 // indirect + github.com/FilenCloudDienste/filen-sdk-go v0.0.39 // indirect + github.com/Files-com/files-sdk-go/v3 v3.3.194 // indirect + github.com/IBM/go-sdk-core/v5 v5.23.1 // indirect + github.com/Max-Sum/base32768 v0.0.0-20230304063302-18e6ce5945fd // indirect + github.com/Microsoft/go-winio v0.6.2 // indirect + github.com/ProtonMail/bcrypt v0.0.0-20211005172633-e235017c1baf // indirect + github.com/ProtonMail/gluon v0.17.1-0.20230724134000-308be39be96e // indirect + github.com/ProtonMail/go-crypto v1.4.1 // indirect + github.com/ProtonMail/go-srp v0.0.7 // indirect + github.com/ProtonMail/gopenpgp/v3 v3.4.1 // indirect + github.com/PuerkitoBio/goquery v1.12.0 // indirect + github.com/a1ex3/zstd-seekable-format-go/pkg v0.10.0 // indirect + github.com/abbot/go-http-auth v0.4.0 // indirect + github.com/adrg/xdg v0.5.3 // indirect + github.com/anchore/go-lzo v0.1.1 // indirect + github.com/andybalholm/brotli v1.2.2 // indirect + github.com/andybalholm/cascadia v1.3.4 // indirect + github.com/apache/arrow-go/v18 v18.7.0 // indirect + github.com/appscode/go-querystring v0.0.0-20170504095604-0126cfb3f1dc // indirect + github.com/aws/aws-sdk-go-v2 v1.43.7 // indirect + github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.18 // indirect + github.com/aws/aws-sdk-go-v2/config v1.32.30 // indirect + github.com/aws/aws-sdk-go-v2/credentials v1.19.29 // indirect + github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.30 // indirect + github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.34 // indirect + github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.38 // indirect + github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.38 // indirect + github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.39 // indirect + github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.17 // indirect + github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.31 // indirect + github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.38 // indirect + github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.39 // indirect + github.com/aws/aws-sdk-go-v2/service/s3 v1.107.3 // indirect + github.com/aws/aws-sdk-go-v2/service/signin v1.4.1 // indirect + github.com/aws/aws-sdk-go-v2/service/sso v1.32.1 // indirect + github.com/aws/aws-sdk-go-v2/service/ssooidc v1.37.1 // indirect + github.com/aws/aws-sdk-go-v2/service/sts v1.44.1 // indirect + github.com/aws/smithy-go v1.27.8 // indirect + github.com/bahlo/generic-list-go v0.2.0 // indirect + github.com/beevik/ntp v1.5.0 // indirect + github.com/beorn7/perks v1.0.1 // indirect + github.com/boombuler/barcode v1.1.0 // indirect + github.com/bradenaw/juniper v0.15.3 // indirect + github.com/buengese/sgzip v0.1.1 // indirect + github.com/buger/jsonparser v1.2.0 // indirect + github.com/calebcase/tmpfile v1.0.3 // indirect + github.com/cenkalti/backoff/v4 v4.3.0 // indirect + github.com/cespare/xxhash/v2 v2.3.0 // indirect + github.com/chilts/sid v0.0.0-20190607042430-660e94789ec9 // indirect + github.com/clipperhouse/uax29/v2 v2.7.0 // indirect + github.com/cloudflare/circl v1.6.4 // indirect + github.com/cloudinary/cloudinary-go/v2 v2.16.0 // indirect + github.com/cloudsoda/go-smb2 v0.0.0-20260701064823-d8c5600d73b8 // indirect + github.com/cloudsoda/sddl v0.0.0-20250224235906-926454e91efc // indirect + github.com/coder/websocket v1.8.15 // indirect + github.com/colinmarc/hdfs/v2 v2.4.0 // indirect + github.com/coreos/go-semver v0.3.1 // indirect + github.com/coreos/go-systemd/v22 v22.6.0 // indirect + github.com/creasty/defaults v1.8.0 // indirect + github.com/cronokirby/saferith v0.33.1-0.20250226174546-1f11f94ce488 // indirect + github.com/diskfs/go-diskfs v1.9.4 // indirect + github.com/dlclark/regexp2 v1.12.0 // indirect + github.com/dromara/dongle v1.0.1 // indirect + github.com/dropbox/dropbox-sdk-go-unofficial/v6 v6.4.0 // indirect + github.com/ebitengine/purego v0.10.1 // indirect + github.com/emersion/go-message v0.18.2 // indirect + github.com/emersion/go-vcard v0.0.0-20260618161152-d854b7e0e2d3 // indirect + github.com/fatih/color v1.19.0 // indirect + github.com/felixge/httpsnoop v1.1.0 // indirect + github.com/flynn/noise v1.1.0 // indirect + github.com/gabriel-vasile/mimetype v1.4.15 // indirect + github.com/geoffgarside/ber v1.2.0 // indirect + github.com/ghodss/yaml v1.0.0 // indirect + github.com/go-chi/chi/v5 v5.3.1 // indirect + github.com/go-darwin/apfs v0.0.0-20211011131704-f84b94dbf348 // indirect + github.com/go-faster/errors v0.8.0 // indirect + github.com/go-faster/jx v1.2.0 // indirect + github.com/go-faster/xor v1.0.0 // indirect + github.com/go-faster/yaml v0.4.6 // indirect + github.com/go-git/go-billy/v5 v5.9.0 // indirect + github.com/go-logr/logr v1.4.3 // indirect + github.com/go-logr/stdr v1.2.2 // indirect + github.com/go-ole/go-ole v1.3.0 // indirect + github.com/go-openapi/errors v0.22.8 // indirect + github.com/go-openapi/strfmt v0.27.0 // indirect + github.com/go-playground/locales v0.14.1 // indirect + github.com/go-playground/universal-translator v0.18.1 // indirect + github.com/go-playground/validator/v10 v10.30.3 // indirect + github.com/go-resty/resty/v2 v2.17.2 // indirect + github.com/go-viper/mapstructure/v2 v2.5.0 // indirect + github.com/goccy/go-json v0.10.6 // indirect + github.com/gofrs/flock v0.13.0 // indirect + github.com/gogo/protobuf v1.3.2 // indirect + github.com/golang-jwt/jwt/v4 v4.5.2 // indirect + github.com/golang-jwt/jwt/v5 v5.3.1 // indirect + github.com/google/btree v1.1.3 // indirect + github.com/google/flatbuffers v25.12.19+incompatible // indirect + github.com/google/s2a-go v0.1.9 // indirect + github.com/google/uuid v1.6.0 // indirect + github.com/googleapis/enterprise-certificate-proxy v0.3.18 // indirect + github.com/googleapis/gax-go/v2 v2.22.0 // indirect + github.com/gorilla/schema v1.4.1 // indirect + github.com/gotd/contrib v0.20.0 // indirect + github.com/gotd/ige v0.3.0 // indirect + github.com/gotd/log v0.1.0 // indirect + github.com/gotd/neo v0.1.5 // indirect + github.com/hashicorp/errwrap v1.1.0 // indirect + github.com/hashicorp/go-cleanhttp v0.5.2 // indirect + github.com/hashicorp/go-multierror v1.1.1 // indirect + github.com/hashicorp/go-retryablehttp v0.7.8 // indirect + github.com/hashicorp/go-uuid v1.0.3 // indirect + github.com/internxt/rclone-adapter v0.0.0-20260708165336-dd6561bacfa2 // indirect + github.com/iyear/connectproxy v0.1.1 // indirect + github.com/jcmturner/aescts/v2 v2.0.0 // indirect + github.com/jcmturner/dnsutils/v2 v2.0.0 // indirect + github.com/jcmturner/gofork v1.7.6 // indirect + github.com/jcmturner/goidentity/v6 v6.0.1 // indirect + github.com/jcmturner/gokrb5/v8 v8.4.4 // indirect + github.com/jcmturner/rpc/v2 v2.0.3 // indirect + github.com/jlaffaye/ftp v0.2.1-0.20251026020404-6602e981a1bb // indirect + github.com/jtolds/gls v4.20.0+incompatible // indirect + github.com/jtolio/noiseconn v0.0.0-20231127013910-f6d9ecbf1de7 // indirect + github.com/jzelinskie/whirlpool v0.0.0-20201016144138-0675e54bb004 // indirect + github.com/klauspost/compress v1.19.2 // indirect + github.com/klauspost/cpuid/v2 v2.4.0 // indirect + github.com/koofr/go-httpclient v0.0.0-20240520111329-e20f8f203988 // indirect + github.com/koofr/go-koofrclient v0.0.0-20221207135200-cbd7fc9ad6a6 // indirect + github.com/kr/fs v0.1.0 // indirect + github.com/kylelemons/godebug v1.1.0 // indirect + github.com/lanrat/extsort v1.4.2 // indirect + github.com/leodido/go-urn v1.4.0 // indirect + github.com/lpar/calendar v0.2.0 // indirect + github.com/lufia/plan9stats v0.0.0-20260627054121-477a66015f15 // indirect + github.com/mailru/easyjson v0.9.2 // indirect + github.com/mattn/go-colorable v0.1.15 // indirect + github.com/mattn/go-isatty v0.0.23 // indirect + github.com/mattn/go-runewidth v0.0.24 // indirect + github.com/mitchellh/go-homedir v1.1.0 // indirect + github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 // indirect + github.com/ncw/swift/v2 v2.0.5 // indirect + github.com/ogen-go/ogen v1.23.0 // indirect + github.com/oklog/ulid/v2 v2.1.1 // indirect + github.com/oracle/oci-go-sdk/v65 v65.121.0 // indirect + github.com/panjf2000/ants/v2 v2.12.1 // indirect + github.com/patrickmn/go-cache v2.1.0+incompatible // indirect + github.com/pengsrc/go-shared v0.2.1-0.20190131101655-1999055a4a14 // indirect + github.com/peterh/liner v1.2.2 // indirect + github.com/pierrec/lz4/v4 v4.1.27 // indirect + github.com/pkg/browser v0.0.0-20240102092130-5ac0b6a4141c // indirect + github.com/pkg/errors v0.9.1 // indirect + github.com/pkg/sftp v1.13.11 // indirect + github.com/pkg/xattr v0.4.12 // indirect + github.com/power-devops/perfstat v0.0.0-20240221224432-82ca36839d55 // indirect + github.com/pquerna/otp v1.5.0 // indirect + github.com/prometheus/client_golang v1.23.2 // indirect + github.com/prometheus/client_model v0.6.2 // indirect + github.com/prometheus/common v0.70.0 // indirect + github.com/prometheus/procfs v0.21.1 // indirect + github.com/putdotio/go-putio/putio v0.0.0-20200123120452-16d982cac2b8 // indirect + github.com/rclone/Proton-API-Bridge v1.0.5 // indirect + github.com/rclone/go-proton-api v1.0.4 // indirect + github.com/refraction-networking/utls v1.8.2 // indirect + github.com/relvacode/iso8601 v1.7.0 // indirect + github.com/rfjakob/eme v1.2.0 // indirect + github.com/sabhiram/go-gitignore v0.0.0-20210923224102-525f6e181f06 // indirect + github.com/samber/lo v1.53.0 // indirect + github.com/segmentio/asm v1.2.1 // indirect + github.com/shirou/gopsutil/v4 v4.26.6 // indirect + github.com/shopspring/decimal v1.4.0 // indirect + github.com/sirupsen/logrus v1.9.4 // indirect + github.com/skratchdot/open-golang v0.0.0-20200116055534-eef842397966 // indirect + github.com/smarty/assertions v1.16.0 // indirect + github.com/sony/gobreaker/v2 v2.4.0 // indirect + github.com/spacemonkeygo/monkit/v3 v3.0.25-0.20251022131615-eb24eb109368 // indirect + github.com/spf13/pflag v1.0.10 // indirect + github.com/stretchr/testify v1.12.1 // indirect + github.com/t3rm1n4l/go-mega v0.0.0-20260717075258-c6acd6a5bd04 // indirect + github.com/tklauser/go-sysconf v0.4.0 // indirect + github.com/tklauser/numcpus v0.12.0 // indirect + github.com/tyler-smith/go-bip39 v1.1.0 // indirect + github.com/ulikunitz/xz v0.5.15 // indirect + github.com/unknwon/goconfig v1.0.0 // indirect + github.com/wk8/go-ordered-map/v2 v2.1.8 // indirect + github.com/xanzy/ssh-agent v0.3.3 // indirect + github.com/youmark/pkcs8 v0.0.0-20240726163527-a2c0da244d78 // indirect + github.com/yuin/goldmark v1.8.4 // indirect + github.com/yunify/qingstor-sdk-go/v3 v3.2.0 // indirect + github.com/yusufpapurcu/wmi v1.2.4 // indirect + github.com/zeebo/blake3 v0.2.4 // indirect + github.com/zeebo/errs v1.4.0 // indirect + github.com/zeebo/xxh3 v1.1.0 // indirect + go.opentelemetry.io/auto/sdk v1.2.1 // indirect + go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.69.0 // indirect + go.opentelemetry.io/otel v1.44.0 // indirect + go.opentelemetry.io/otel/metric v1.44.0 // indirect + go.opentelemetry.io/otel/trace v1.44.0 // indirect + go.uber.org/atomic v1.11.0 // indirect + go.uber.org/multierr v1.11.0 // indirect + go.uber.org/zap v1.28.0 // indirect + go.yaml.in/yaml/v2 v2.4.4 // indirect + go.yaml.in/yaml/v3 v3.0.5 // indirect + golang.org/x/crypto v0.56.0 // indirect + golang.org/x/exp v0.0.0-20260709172345-9ea1abe57597 // indirect + golang.org/x/image v0.45.0 // indirect + golang.org/x/mod v0.38.0 // indirect + golang.org/x/net v0.58.0 // indirect + golang.org/x/oauth2 v0.36.0 // indirect + golang.org/x/sync v0.22.0 // indirect + golang.org/x/sys v0.47.0 // indirect + golang.org/x/term v0.45.0 // indirect + golang.org/x/text v0.41.0 // indirect + golang.org/x/time v0.15.0 // indirect + golang.org/x/tools v0.48.0 // indirect + google.golang.org/api v0.279.0 // indirect + google.golang.org/genproto/googleapis/rpc v0.0.0-20260715232425-e75dac1f907d // indirect + google.golang.org/grpc v1.84.0-dev.0.20260723093437-b6eac429d7b6 // indirect + google.golang.org/protobuf v1.36.11 // indirect + gopkg.in/natefinch/lumberjack.v2 v2.2.1 // indirect + gopkg.in/validator.v2 v2.0.1 // indirect + gopkg.in/yaml.v2 v2.4.0 // indirect + gopkg.in/yaml.v3 v3.0.1 // indirect + moul.io/http2curl/v2 v2.3.0 // indirect + rsc.io/qr v0.2.0 // indirect + sigs.k8s.io/yaml v1.6.0 // indirect + storj.io/common v0.0.0-20260629224719-ba1bff0a7846 // indirect + storj.io/drpc v1.0.0 // indirect + storj.io/eventkit v0.0.0-20260716074419-6861a92e2aa5 // indirect + storj.io/infectious v0.0.2 // indirect + storj.io/picobuf v0.0.4 // indirect + storj.io/uplink v1.14.3 // indirect +) diff --git a/go.sum b/go.sum new file mode 100644 index 0000000..9a73884 --- /dev/null +++ b/go.sum @@ -0,0 +1,737 @@ +cloud.google.com/go/auth v0.20.0 h1:kXTssoVb4azsVDoUiF8KvxAqrsQcQtB53DcSgta74CA= +cloud.google.com/go/auth v0.20.0/go.mod h1:942/yi/itH1SsmpyrbnTMDgGfdy2BUqIKyd0cyYLc5Q= +cloud.google.com/go/auth/oauth2adapt v0.2.8 h1:keo8NaayQZ6wimpNSmW5OPc283g65QNIiLpZnkHRbnc= +cloud.google.com/go/auth/oauth2adapt v0.2.8/go.mod h1:XQ9y31RkqZCcwJWNSx2Xvric3RrU88hAYYbjDWYDL+c= +cloud.google.com/go/compute/metadata v0.9.0 h1:pDUj4QMoPejqq20dK0Pg2N4yG9zIkYGdBtwLoEkH9Zs= +cloud.google.com/go/compute/metadata v0.9.0/go.mod h1:E0bWwX5wTnLPedCKqk3pJmVgCBSM6qQI1yTBdEb3C10= +github.com/Azure/azure-sdk-for-go/sdk/azcore v1.22.0 h1:aokoqcHvaGjiM3VpjKDfMMnF/8epJ+Q1HLJ7CudztqE= +github.com/Azure/azure-sdk-for-go/sdk/azcore v1.22.0/go.mod h1:/WYEx9pcM9Y+Dd/APJaNlSvVSvzl54rrMdZT5+Oi2LM= +github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.14.0 h1:CU4+EJeJi3TKYWEcYuSdWsjzw0nVsK/H0MSQOiPcymU= +github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.14.0/go.mod h1:q0+UTSRvShwUCrR/s5HtyInYphN7Wvxb7snFM3u+SLA= +github.com/Azure/azure-sdk-for-go/sdk/azidentity/cache v0.4.0 h1:xFaZZ+IubdftrDHnGGwZ6QvQ3KHTtWl2MCK+GMt2vxs= +github.com/Azure/azure-sdk-for-go/sdk/azidentity/cache v0.4.0/go.mod h1:mCBhUhlMjLLJKr5aqw2TNS/VqJOie8MzWq3DAMJeKso= +github.com/Azure/azure-sdk-for-go/sdk/internal v1.12.0 h1:fhqpLE3UEXi9lPaBRpQ6XuRW0nU7hgg4zlmZZa+a9q4= +github.com/Azure/azure-sdk-for-go/sdk/internal v1.12.0/go.mod h1:7dCRMLwisfRH3dBupKeNCioWYUZ4SS09Z14H+7i8ZoY= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/storage/armstorage v1.8.1 h1:/Zt+cDPnpC3OVDm/JKLOs7M2DKmLRIIp3XIx9pHHiig= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/storage/armstorage v1.8.1/go.mod h1:Ng3urmn6dYe8gnbCMoHHVl5APYz2txho3koEkV2o2HA= +github.com/Azure/azure-sdk-for-go/sdk/storage/azblob v1.8.0 h1:irsmOWwkp0KCTTNS5e2hdFeIvSQClQo2No3IaNmL3Vw= +github.com/Azure/azure-sdk-for-go/sdk/storage/azblob v1.8.0/go.mod h1:GWcBkQj3MqN7ozHKLaCCAuNLiXoIGv2RtanfAwSjY/Y= +github.com/Azure/azure-sdk-for-go/sdk/storage/azfile v1.7.0 h1:cuiKf1UVyWHu+XSQghPZR/qEF43JIcuk2CDqMlPiT6M= +github.com/Azure/azure-sdk-for-go/sdk/storage/azfile v1.7.0/go.mod h1:9JSyvgXLPAOC4jfhgZg58XeU2FJHsGmxuYSclCFQ4ZY= +github.com/Azure/go-ntlmssp v0.1.1 h1:l+FM/EEMb0U9QZE7mKNEDw5Mu3mFiaa2GKOoTSsNDPw= +github.com/Azure/go-ntlmssp v0.1.1/go.mod h1:NYqdhxd/8aAct/s4qSYZEerdPuH1liG2/X9DiVTbhpk= +github.com/AzureAD/microsoft-authentication-extensions-for-go/cache v0.1.1 h1:WJTmL004Abzc5wDB5VtZG2PJk5ndYDgVacGqfirKxjM= +github.com/AzureAD/microsoft-authentication-extensions-for-go/cache v0.1.1/go.mod h1:tCcJZ0uHAmvjsVYzEFivsRTN00oz5BEsRgQHu5JZ9WE= +github.com/AzureAD/microsoft-authentication-library-for-go v1.7.2 h1:RHK7bS+HQMslb1sZpAokUt+zTVmue0hKSs2C791hhzU= +github.com/AzureAD/microsoft-authentication-library-for-go v1.7.2/go.mod h1:HKpQxkWaGLJ+D/5H8QRpyQXA1eKjxkFlOMwck5+33Jk= +github.com/FilenCloudDienste/filen-sdk-go v0.0.39 h1:tgV5jYL6dsXop9TpDTIQU6UwJjws122HrwskaEE/igY= +github.com/FilenCloudDienste/filen-sdk-go v0.0.39/go.mod h1:0cBhKXQg49XbKZZfk5TCDa3sVLP+xMxZTWL+7KY0XR0= +github.com/Files-com/files-sdk-go/v3 v3.3.194 h1:dtOFxSTWWRpkmvXa6ycNiw8dVDu1wkgzcXyVV1VafNc= +github.com/Files-com/files-sdk-go/v3 v3.3.194/go.mod h1:rl0WumSN9gSo775DgvQv+wMQ8rlb0ES/1hU5jkMtLXg= +github.com/IBM/go-sdk-core/v5 v5.23.1 h1:fxLusCG+GlGY1SM7BYAAiiSFhzKUs5zqs0P8knDqeFU= +github.com/IBM/go-sdk-core/v5 v5.23.1/go.mod h1:yO+OQpByKDLTvpEcsFFexgzpeR8eRfCFWAYzxkAu4bk= +github.com/Masterminds/semver/v3 v3.2.0 h1:3MEsd0SM6jqZojhjLWWeBY+Kcjy9i6MQAeY7YgDP83g= +github.com/Masterminds/semver/v3 v3.2.0/go.mod h1:qvl/7zhW3nngYb5+80sSMF+FG2BjYrf8m9wsX0PNOMQ= +github.com/Max-Sum/base32768 v0.0.0-20230304063302-18e6ce5945fd h1:nzE1YQBdx1bq9IlZinHa+HVffy+NmVRoKr+wHN8fpLE= +github.com/Max-Sum/base32768 v0.0.0-20230304063302-18e6ce5945fd/go.mod h1:C8yoIfvESpM3GD07OCHU7fqI7lhwyZ2Td1rbNbTAhnc= +github.com/Microsoft/go-winio v0.5.2/go.mod h1:WpS1mjBmmwHBEWmogvA2mj8546UReBk4v8QkMxJ6pZY= +github.com/Microsoft/go-winio v0.6.2 h1:F2VQgta7ecxGYO8k3ZZz3RS8fVIXVxONVUPlNERoyfY= +github.com/Microsoft/go-winio v0.6.2/go.mod h1:yd8OoFMLzJbo9gZq8j5qaps8bJ9aShtEA8Ipt1oGCvU= +github.com/ProtonMail/bcrypt v0.0.0-20210511135022-227b4adcab57/go.mod h1:HecWFHognK8GfRDGnFQbW/LiV7A3MX3gZVs45vk5h8I= +github.com/ProtonMail/bcrypt v0.0.0-20211005172633-e235017c1baf h1:yc9daCCYUefEs69zUkSzubzjBbL+cmOXgnmt9Fyd9ug= +github.com/ProtonMail/bcrypt v0.0.0-20211005172633-e235017c1baf/go.mod h1:o0ESU9p83twszAU8LBeJKFAAMX14tISa0yk4Oo5TOqo= +github.com/ProtonMail/gluon v0.17.1-0.20230724134000-308be39be96e h1:lCsqUUACrcMC83lg5rTo9Y0PnPItE61JSfvMyIcANwk= +github.com/ProtonMail/gluon v0.17.1-0.20230724134000-308be39be96e/go.mod h1:Og5/Dz1MiGpCJn51XujZwxiLG7WzvvjE5PRpZBQmAHo= +github.com/ProtonMail/go-crypto v0.0.0-20230321155629-9a39f2531310/go.mod h1:8TI4H3IbrackdNgv+92dI+rhpCaLqM0IfpgCgenFvRE= +github.com/ProtonMail/go-crypto v1.4.1 h1:9RfcZHqEQUvP8RzecWEUafnZVtEvrBVL9BiF67IQOfM= +github.com/ProtonMail/go-crypto v1.4.1/go.mod h1:e1OaTyu5SYVrO9gKOEhTc+5UcXtTUa+P3uLudwcgPqo= +github.com/ProtonMail/go-srp v0.0.7 h1:Sos3Qk+th4tQR64vsxGIxYpN3rdnG9Wf9K4ZloC1JrI= +github.com/ProtonMail/go-srp v0.0.7/go.mod h1:giCp+7qRnMIcCvI6V6U3S1lDDXDQYx2ewJ6F/9wdlJk= +github.com/ProtonMail/gopenpgp/v3 v3.4.1 h1:K7uUhSHSJxORZ+RuHpilTT6S4MA2whCRlXNwLqd0+ys= +github.com/ProtonMail/gopenpgp/v3 v3.4.1/go.mod h1:bGdV9f6edhmd581wzXsQCTKdH8bXBbyhkgDKPjwPc6U= +github.com/PuerkitoBio/goquery v1.12.0 h1:pAcL4g3WRXekcB9AU/y1mbKez2dbY2AajVhtkO8RIBo= +github.com/PuerkitoBio/goquery v1.12.0/go.mod h1:802ej+gV2y7bbIhOIoPY5sT183ZW0YFofScC4q/hIpQ= +github.com/a1ex3/zstd-seekable-format-go/pkg v0.10.0 h1:iLDOF0rdGTrol/q8OfPIIs5kLD8XvA2q75o6Uq/tgak= +github.com/a1ex3/zstd-seekable-format-go/pkg v0.10.0/go.mod h1:DrEWcQJjz7t5iF2duaiyhg4jyoF0kxOD6LtECNGkZ/Q= +github.com/aalpar/deheap v1.1.2 h1:MABHLcnjqsffb8GLkUFDigqpBBxOMz0DoKM9QfELeTw= +github.com/aalpar/deheap v1.1.2/go.mod h1:A+nfkD4JbS05sewV0he/MYgR/90vfqyMoNNROgs+rmA= +github.com/abbot/go-http-auth v0.4.0 h1:QjmvZ5gSC7jm3Zg54DqWE/T5m1t2AfDu6QlXJT0EVT0= +github.com/abbot/go-http-auth v0.4.0/go.mod h1:Cz6ARTIzApMJDzh5bRMSUou6UMSp0IEXg9km/ci7TJM= +github.com/adrg/xdg v0.5.3 h1:xRnxJXne7+oWDatRhR1JLnvuccuIeCoBu2rtuLqQB78= +github.com/adrg/xdg v0.5.3/go.mod h1:nlTsY+NNiCBGCK2tpm09vRqfVzrc2fLmXGpBLF0zlTQ= +github.com/anchore/go-lzo v0.1.1 h1:IwL/fvkdtlIrYIXck6WxZ3nb8WjjHziYYmGxlooyOnM= +github.com/anchore/go-lzo v0.1.1/go.mod h1:3kLx0bve2oN1iDwgM1U5zGku1Tfbdb0No5qp1eL1fIk= +github.com/andybalholm/brotli v1.2.2 h1:HzTuoo2ErYQqf5qvcJInB8uvqSVxRttzkFexPWtnceM= +github.com/andybalholm/brotli v1.2.2/go.mod h1:rzTDkvFWvIrjDXZHkuS16NPggd91W3kUSvPlQ1pLaKY= +github.com/andybalholm/cascadia v1.3.4 h1:vM2lgh0Vru9Vwyfm4cQqWP2HHMW0u0+2PAW7Q38Qufg= +github.com/andybalholm/cascadia v1.3.4/go.mod h1:BLRmbRjpEtNKieZOCCvYj4RqN+KRA41GBe/5O+G93kM= +github.com/apache/arrow-go/v18 v18.7.0 h1:Vw/i+cJyebUofT7JlqFpe65LrmwxULn166jjwStM4HY= +github.com/apache/arrow-go/v18 v18.7.0/go.mod h1:PM6IigLJkdMwIpeHXnymo+xZ52f42a9EYiLtRel4p/A= +github.com/apache/thrift v0.24.0 h1:zy31L1a49QTNB2bG1BBfMXol3yJrTH975G3pPubQVLQ= +github.com/apache/thrift v0.24.0/go.mod h1:zPt6WxgvTOM6hF92y8C+MkEM5LMxZuk4JcQOiU4Esvs= +github.com/appscode/go-querystring v0.0.0-20170504095604-0126cfb3f1dc h1:LoL75er+LKDHDUfU5tRvFwxH0LjPpZN8OoG8Ll+liGU= +github.com/appscode/go-querystring v0.0.0-20170504095604-0126cfb3f1dc/go.mod h1:w648aMHEgFYS6xb0KVMMtZ2uMeemhiKCuD2vj6gY52A= +github.com/aws/aws-sdk-go-v2 v1.43.7 h1:msCzvkeYJA9ehbV8mRRmkZLo/zJg/+yDVLNtflg83hQ= +github.com/aws/aws-sdk-go-v2 v1.43.7/go.mod h1:tXpPM+v0D1lndmga+HqqLDIzUFJlEeR21aspVklHF00= +github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.18 h1:LAfOuhAH331fmOjTQpAaOlH+Ftn7RzSDJ2VFwjdMMy4= +github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.18/go.mod h1:4e5xhuXHx1e4U9EthvbPP1r/DIMp5c2823OL8karzcM= +github.com/aws/aws-sdk-go-v2/config v1.32.30 h1:XwsEzpTJfQYJbFicz/QMLwAZdyeNVVoOEkbF7R3gPJk= +github.com/aws/aws-sdk-go-v2/config v1.32.30/go.mod h1:Ud32SuMc+/9BGxfpSVld7HrE2o05JwKmXY4M3jOQNZU= +github.com/aws/aws-sdk-go-v2/credentials v1.19.29 h1:WHZGssHH887cO0ox07SIQZsFx3MKD4ps6w0xUEmnKYQ= +github.com/aws/aws-sdk-go-v2/credentials v1.19.29/go.mod h1:Mhl0xR6zjguiuj00XRx2wMx22sAltk7oya39sT7fdg8= +github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.30 h1:/hi1JADLEW9YYryEz1w4GQu0EtP23pP553Cf9KgsDV4= +github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.30/go.mod h1:/3AOgy4K17Dm4ucMZVC/MJkzy5kmfKUcINRHZyo0koQ= +github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.34 h1:Pn7OsMwBLbkZ6OnCxWHAjf0L/22H8cnhxZC0uPwtMtg= +github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.34/go.mod h1:eToXR/Gk1uqpn04eSmdgVXwfS0WvH8aG4eBFr8ygbpU= +github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.38 h1:MBMg0zJ6i4TkAJ0dVFLKKn2cOkY6FkicmUDM67BRr6g= +github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.38/go.mod h1:9MWuJbyiUyj6eA7W1/zm1zuePDPSB3g+xcgRQeMWsXc= +github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.38 h1:lHm4jPf3k1Lz5ZWc+Vcn3MKVwym+26kWCba9FkJ4f0Y= +github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.38/go.mod h1:Rn+P2XR+FbyZzjmWKjg/KUZNxmGfr5oZwh5jQiE+CzI= +github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.39 h1:vo4xvMRs/F6h1E52qsgLqCQgWIQXgIJUauG6rlZEh4U= +github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.39/go.mod h1:jB03R1ij/A+OE2e1dz6vgj076gd7vlYcfstAzj3HcnU= +github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.17 h1:OvYZOB3qA6zvfdRFiRFRzVSiElMYrz3GdntkXZxlp1o= +github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.17/go.mod h1:JgR/2Ew50ACfIWau1oeMRX59tMtC0kM+PYQGEaT04cY= +github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.31 h1:uZOinZb+h7lZw8IYzP1z1IuEnueB76/EFkcf/fEW4Ag= +github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.31/go.mod h1:NRtwAM/p5VRt03TlEUs0pH3TeWamWdf4YyJpSrzPYLc= +github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.38 h1:H/5TI1jqaHsNoDQ60UwvPvJBg4GURkinXI3Qga29t2w= +github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.38/go.mod h1:PTVFf+XH++7NJOky+RLBYQx0QA5NcaeEYFQ2fsi0nwo= +github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.39 h1:HLPAVrlLDaN2boN0xJx7MgaQDNEO3Q+c9L6kl/8m47Q= +github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.39/go.mod h1:Pg/dVfsNkm1hsIDK/gMvCKtmyNfNTV12mrgHqVE/6Oo= +github.com/aws/aws-sdk-go-v2/service/s3 v1.107.3 h1:IKoCZqfWfZzSBi16QFQ+QcbQ3LRQ7QgB1S5tDAyPBQQ= +github.com/aws/aws-sdk-go-v2/service/s3 v1.107.3/go.mod h1:RBpRcXiM4s2pOInVs32GsBonnje+fiAj4mcrStRmlCA= +github.com/aws/aws-sdk-go-v2/service/signin v1.4.1 h1:V7ZZ300WPXGjvkyore5DGe0ljVPOxCXie/thWdtSBXE= +github.com/aws/aws-sdk-go-v2/service/signin v1.4.1/go.mod h1:mxC0nT/C8wMMS97DemZPzvUZxvIt+2Iq+eS3JdFZGgg= +github.com/aws/aws-sdk-go-v2/service/sso v1.32.1 h1:gYFYh4iLLcAOJRLNPY2aD2g9DIhKn4eof8UkIrr1rTk= +github.com/aws/aws-sdk-go-v2/service/sso v1.32.1/go.mod h1:u8af9Nqkmqnr96f7v9nHqzZT9XBwbXEkTiqT4ROuJSE= +github.com/aws/aws-sdk-go-v2/service/ssooidc v1.37.1 h1:arjT9Cm3/WYbGmD5TUZHk4UQn4Lle1fUNZs5FC6CtF0= +github.com/aws/aws-sdk-go-v2/service/ssooidc v1.37.1/go.mod h1:DMPWJBjYs6+3+f/qhBFEFPPlQ6NlhWjai3dJNvipJ84= +github.com/aws/aws-sdk-go-v2/service/sts v1.44.1 h1:RvfHDg+xvAeZ+5741vUEjpOVtYSIm93W2zhx10Xtydw= +github.com/aws/aws-sdk-go-v2/service/sts v1.44.1/go.mod h1:9gdl4RrflIdpDb2TlXshWgR1F9TeCkvqDx77Vpr4Z/Q= +github.com/aws/smithy-go v1.27.8 h1:FR0dxZfIlV7Z8eh2iHfIofdunw382XsDV3Mxt9nUvRY= +github.com/aws/smithy-go v1.27.8/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc= +github.com/bahlo/generic-list-go v0.2.0 h1:5sz/EEAK+ls5wF+NeqDpk5+iNdMDXrh3z3nPnH1Wvgk= +github.com/bahlo/generic-list-go v0.2.0/go.mod h1:2KvAjgMlE5NNynlg/5iLrrCCZ2+5xWbdbCW3pNTGyYg= +github.com/beevik/ntp v1.5.0 h1:y+uj/JjNwlY2JahivxYvtmv4ehfi3h74fAuABB9ZSM4= +github.com/beevik/ntp v1.5.0/go.mod h1:mJEhBrwT76w9D+IfOEGvuzyuudiW9E52U2BaTrMOYow= +github.com/beorn7/perks v1.0.1 h1:VlbKKnNfV8bJzeqoa4cOKqO6bYr3WgKZxO8Z16+hsOM= +github.com/beorn7/perks v1.0.1/go.mod h1:G2ZrVWU2WbWT9wwq4/hrbKbnv/1ERSJQ0ibhJ6rlkpw= +github.com/boombuler/barcode v1.0.1-0.20190219062509-6c824513bacc/go.mod h1:paBWMcWSl3LHKBqUq+rly7CNSldXjb2rDl3JlRe0mD8= +github.com/boombuler/barcode v1.1.0 h1:ChaYjBR63fr4LFyGn8E8nt7dBSt3MiU3zMOZqFvVkHo= +github.com/boombuler/barcode v1.1.0/go.mod h1:paBWMcWSl3LHKBqUq+rly7CNSldXjb2rDl3JlRe0mD8= +github.com/bradenaw/juniper v0.15.3 h1:RHIAMEDTpvmzV1wg1jMAHGOoI2oJUSPx3lxRldXnFGo= +github.com/bradenaw/juniper v0.15.3/go.mod h1:UX4FX57kVSaDp4TPqvSjkAAewmRFAfXf27BOs5z9dq8= +github.com/buengese/sgzip v0.1.1 h1:ry+T8l1mlmiWEsDrH/YHZnCVWD2S3im1KLsyO+8ZmTU= +github.com/buengese/sgzip v0.1.1/go.mod h1:i5ZiXGF3fhV7gL1xaRRL1nDnmpNj0X061FQzOS8VMas= +github.com/buger/jsonparser v1.2.0 h1:4EFcvK1kD4jyj6YqNK6skK6w+y7FHHBR+XBCtxwu/6g= +github.com/buger/jsonparser v1.2.0/go.mod h1:6RYKKt7H4d4+iWqouImQ9R2FZql3VbhNgx27UK13J/0= +github.com/bwesterb/go-ristretto v1.2.0/go.mod h1:fUIoIZaG73pV5biE2Blr2xEzDoMj7NFEuV9ekS419A0= +github.com/bytedance/sonic v1.13.2 h1:8/H1FempDZqC4VqjptGo14QQlJx8VdZJegxs6wwfqpQ= +github.com/bytedance/sonic v1.13.2/go.mod h1:o68xyaF9u2gvVBuGHPlUVCy+ZfmNNO5ETf1+KgkJhz4= +github.com/bytedance/sonic/loader v0.2.4 h1:ZWCw4stuXUsn1/+zQDqeE7JKP+QO47tz7QCNan80NzY= +github.com/bytedance/sonic/loader v0.2.4/go.mod h1:N8A3vUdtUebEY2/VQC0MyhYeKUFosQU6FxH2JmUe6VI= +github.com/calebcase/tmpfile v1.0.3 h1:BZrOWZ79gJqQ3XbAQlihYZf/YCV0H4KPIdM5K5oMpJo= +github.com/calebcase/tmpfile v1.0.3/go.mod h1:UAUc01aHeC+pudPagY/lWvt2qS9ZO5Zzof6/tIUzqeI= +github.com/cenkalti/backoff/v4 v4.3.0 h1:MyRJ/UdXutAwSAT+s3wNd7MfTIcy71VQueUuFK343L8= +github.com/cenkalti/backoff/v4 v4.3.0/go.mod h1:Y3VNntkOUPxTVeUxJ/G5vcM//AlwfmyYozVcomhLiZE= +github.com/cespare/xxhash/v2 v2.3.0 h1:UL815xU9SqsFlibzuggzjXhog7bL6oX9BbNZnL2UFvs= +github.com/cespare/xxhash/v2 v2.3.0/go.mod h1:VGX0DQ3Q6kWi7AoAeZDth3/j3BFtOZR5XLFGgcrjCOs= +github.com/chilts/sid v0.0.0-20190607042430-660e94789ec9 h1:z0uK8UQqjMVYzvk4tiiu3obv2B44+XBsvgEJREQfnO8= +github.com/chilts/sid v0.0.0-20190607042430-660e94789ec9/go.mod h1:Jl2neWsQaDanWORdqZ4emBl50J4/aRBBS4FyyG9/PFo= +github.com/clipperhouse/uax29/v2 v2.7.0 h1:+gs4oBZ2gPfVrKPthwbMzWZDaAFPGYK72F0NJv2v7Vk= +github.com/clipperhouse/uax29/v2 v2.7.0/go.mod h1:EFJ2TJMRUaplDxHKj1qAEhCtQPW2tJSwu5BF98AuoVM= +github.com/cloudflare/circl v1.1.0/go.mod h1:prBCrKB9DV4poKZY1l9zBXg2QJY7mvgRvtMxxK7fi4I= +github.com/cloudflare/circl v1.6.4 h1:pOXuDTCEYyzydgUpQ0CQz3LsinKjiSk6nNP5Lt5K64U= +github.com/cloudflare/circl v1.6.4/go.mod h1:YxarevkLlbaHuWsxG6vmYNWBEsSp4pnp7j+4VljMavY= +github.com/cloudinary/cloudinary-go/v2 v2.16.0 h1:0irPbKwRB6V6sdP9+a4P4D5sRUYHXriG5GfTjVq4YBI= +github.com/cloudinary/cloudinary-go/v2 v2.16.0/go.mod h1:ireC4gqVetsjVhYlwjUJwKTbZuWjEIynbR9zQTlqsvo= +github.com/cloudsoda/go-smb2 v0.0.0-20260701064823-d8c5600d73b8 h1:+KC2I+emnT6ccWmo7IHbtchGc/vFqAxUuovo2eUgMuM= +github.com/cloudsoda/go-smb2 v0.0.0-20260701064823-d8c5600d73b8/go.mod h1:1pQXB0vAlzRlqcY7LYKOOZMw0wKfJPFxTLsJRF2Gswo= +github.com/cloudsoda/sddl v0.0.0-20250224235906-926454e91efc h1:0xCWmFKBmarCqqqLeM7jFBSw/Or81UEElFqO8MY+GDs= +github.com/cloudsoda/sddl v0.0.0-20250224235906-926454e91efc/go.mod h1:uvR42Hb/t52HQd7x5/ZLzZEK8oihrFpgnodIJ1vte2E= +github.com/cloudwego/base64x v0.1.5 h1:XPciSp1xaq2VCSt6lF0phncD4koWyULpl5bUxbfCyP4= +github.com/cloudwego/base64x v0.1.5/go.mod h1:0zlkT4Wn5C6NdauXdJRhSKRlJvmclQ1hhJgA0rcu/8w= +github.com/coder/websocket v1.8.15 h1:6B2JPeOGlpff2Uz6vOEH1Vzpi0iUz20A+lPVhPHtNUA= +github.com/coder/websocket v1.8.15/go.mod h1:NX3SzP+inril6yawo5CQXx8+fk145lPDC6pumgx0mVg= +github.com/colinmarc/hdfs/v2 v2.4.0 h1:v6R8oBx/Wu9fHpdPoJJjpGSUxo8NhHIwrwsfhFvU9W0= +github.com/colinmarc/hdfs/v2 v2.4.0/go.mod h1:0NAO+/3knbMx6+5pCv+Hcbaz4xn/Zzbn9+WIib2rKVI= +github.com/coreos/go-semver v0.3.1 h1:yi21YpKnrx1gt5R+la8n5WgS0kCrsPp33dmEyHReZr4= +github.com/coreos/go-semver v0.3.1/go.mod h1:irMmmIw/7yzSRPWryHsK7EYSg09caPQL03VsM8rvUec= +github.com/coreos/go-systemd/v22 v22.6.0 h1:aGVa/v8B7hpb0TKl0MWoAavPDmHvobFe5R5zn0bCJWo= +github.com/coreos/go-systemd/v22 v22.6.0/go.mod h1:iG+pp635Fo7ZmV/j14KUcmEyWF+0X7Lua8rrTWzYgWU= +github.com/creasty/defaults v1.8.0 h1:z27FJxCAa0JKt3utc0sCImAEb+spPucmKoOdLHvHYKk= +github.com/creasty/defaults v1.8.0/go.mod h1:iGzKe6pbEHnpMPtfDXZEr0NVxWnPTjb1bbDy08fPzYM= +github.com/cronokirby/saferith v0.33.0/go.mod h1:QKJhjoqUtBsXCAVEjw38mFqoi7DebT7kthcD7UzbnoA= +github.com/cronokirby/saferith v0.33.1-0.20250226174546-1f11f94ce488 h1:tLWBZgPg6TV67oe76W4p+aUQEWIa52wbcuiz8GFd3vo= +github.com/cronokirby/saferith v0.33.1-0.20250226174546-1f11f94ce488/go.mod h1:QKJhjoqUtBsXCAVEjw38mFqoi7DebT7kthcD7UzbnoA= +github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= +github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= +github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc h1:U9qPSI2PIWSS1VwoXQT9A3Wy9MM3WgvqSxFWenqJduM= +github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= +github.com/diskfs/go-diskfs v1.9.4 h1:0j2d7eG4IjyxL6+ChWbDPocdBCF6HQ4HBWU2WDYWVnc= +github.com/diskfs/go-diskfs v1.9.4/go.mod h1:TePJORO83Adh5pb2SqsxAwaP0fofFxKLkxctiS/9OQc= +github.com/djherbis/times v1.6.0 h1:w2ctJ92J8fBvWPxugmXIv7Nz7Q3iDMKNx9v5ocVH20c= +github.com/djherbis/times v1.6.0/go.mod h1:gOHeRAz2h+VJNZ5Gmc/o7iD9k4wW7NMVqieYCY99oc0= +github.com/dlclark/regexp2 v1.12.0 h1:0j4c5qQmnC6XOWNjP3PIXURXN2gWx76rd3KvgdPkCz8= +github.com/dlclark/regexp2 v1.12.0/go.mod h1:DHkYz0B9wPfa6wondMfaivmHpzrQ3v9q8cnmRbL6yW8= +github.com/dnaeon/go-vcr v1.2.0 h1:zHCHvJYTMh1N7xnV7zf1m1GPBF9Ad0Jk/whtQ1663qI= +github.com/dnaeon/go-vcr v1.2.0/go.mod h1:R4UdLID7HZT3taECzJs4YgbbH6PIGXB6W/sc5OLb6RQ= +github.com/dromara/dongle v1.0.1 h1:si/7UP/EXxnFVZok1cNos70GiMGxInAYMilHQFP5dJs= +github.com/dromara/dongle v1.0.1/go.mod h1:ebFhTaDgxaDIKppycENTWlBsxz8mWCPWOLnsEgDpMv4= +github.com/dropbox/dropbox-sdk-go-unofficial/v6 v6.4.0 h1:OYMx56y2as2FIM6QuS5HGXC2AYtf5xv/2n6INaluF4Y= +github.com/dropbox/dropbox-sdk-go-unofficial/v6 v6.4.0/go.mod h1:gDXhl0OElhzYoDsYWHr1RXjpxjGeLzEvjYzH7sZV73k= +github.com/dsnet/try v0.0.3 h1:ptR59SsrcFUYbT/FhAbKTV6iLkeD6O18qfIWRml2fqI= +github.com/dsnet/try v0.0.3/go.mod h1:WBM8tRpUmnXXhY1U6/S8dt6UWdHTQ7y8A5YSkRCkq40= +github.com/ebitengine/purego v0.10.1 h1:dewVBCBT2GaMu1SrNTYxQhgQBethzfhiwvZiLGP/qyY= +github.com/ebitengine/purego v0.10.1/go.mod h1:iIjxzd6CiRiOG0UyXP+V1+jWqUXVjPKLAI0mRfJZTmQ= +github.com/elliotwutingfeng/asciiset v0.0.0-20260129054604-cfde2086bc57 h1:x5yxNrq8XffV/OoNUeFPM6hxHVi5OTspSTBxr/9pemg= +github.com/elliotwutingfeng/asciiset v0.0.0-20260129054604-cfde2086bc57/go.mod h1:GLo/8fDswSAniFG+BFIaiSPcK610jyzgEhWYPQwuQdw= +github.com/emersion/go-message v0.18.2 h1:rl55SQdjd9oJcIoQNhubD2Acs1E6IzlZISRTK7x/Lpg= +github.com/emersion/go-message v0.18.2/go.mod h1:XpJyL70LwRvq2a8rVbHXikPgKj8+aI0kGdHlg16ibYA= +github.com/emersion/go-vcard v0.0.0-20260618161152-d854b7e0e2d3 h1:B9YK+Tck5mTccyDhtxBzWyqGYcFxLyB6+noMNW4/VgI= +github.com/emersion/go-vcard v0.0.0-20260618161152-d854b7e0e2d3/go.mod h1:HMJKR5wlh/ziNp+sHEDV2ltblO4JD2+IdDOWtGcQBTM= +github.com/emmansun/gmsm v0.15.5/go.mod h1:2m4jygryohSWkaSduFErgCwQKab5BNjURoFrn2DNwyU= +github.com/fatih/color v1.19.0 h1:Zp3PiM21/9Ld6FzSKyL5c/BULoe/ONr9KlbYVOfG8+w= +github.com/fatih/color v1.19.0/go.mod h1:zNk67I0ZUT1bEGsSGyCZYZNrHuTkJJB+r6Q9VuMi0LE= +github.com/felixge/httpsnoop v1.1.0 h1:3YtUj32ZZkqZtt3sZZsClsymw/QDuVfpNhoA31zeORc= +github.com/felixge/httpsnoop v1.1.0/go.mod h1:Zqxgdd+1Rkcz8euOqdr7lqgCRJztwr5hp9vDSi5UZCE= +github.com/flynn/noise v1.1.0 h1:KjPQoQCEFdZDiP03phOvGi11+SVVhBG2wOWAorLsstg= +github.com/flynn/noise v1.1.0/go.mod h1:xbMo+0i6+IGbYdJhF31t2eR1BIU0CYc12+BNAKwUTag= +github.com/fsnotify/fsnotify v1.7.0 h1:8JEhPFa5W2WU7YfeZzPNqzMP6Lwt7L2715Ggo0nosvA= +github.com/fsnotify/fsnotify v1.7.0/go.mod h1:40Bi/Hjc2AVfZrqy+aj+yEI+/bRxZnMJyTJwOpGvigM= +github.com/gabriel-vasile/mimetype v1.4.15 h1:05iP/CYtZ/w455R/KZM6rZ5ieAdh99UPtd+d3YzLmaI= +github.com/gabriel-vasile/mimetype v1.4.15/go.mod h1:azpTcoLcDZRNgFou5j+APrqQx9HqVPWa6ijYQIIVswQ= +github.com/geoffgarside/ber v1.2.0 h1:/loowoRcs/MWLYmGX9QtIAbA+V/FrnVLsMMPhwiRm64= +github.com/geoffgarside/ber v1.2.0/go.mod h1:jVPKeCbj6MvQZhwLYsGwaGI52oUorHoHKNecGT85ZCc= +github.com/ghodss/yaml v1.0.0 h1:wQHKEahhL6wmXdzwWG11gIVCkOv05bNOh+Rxn0yngAk= +github.com/ghodss/yaml v1.0.0/go.mod h1:4dBDuWmgqj2HViK6kFavaiC9ZROes6MMH2rRYeMEF04= +github.com/gin-contrib/sse v1.0.0 h1:y3bT1mUWUxDpW4JLQg/HnTqV4rozuW4tC9eFKTxYI9E= +github.com/gin-contrib/sse v1.0.0/go.mod h1:zNuFdwarAygJBht0NTKiSi3jRf6RbqeILZ9Sp6Slhe0= +github.com/gin-gonic/gin v1.10.0 h1:nTuyha1TYqgedzytsKYqna+DfLos46nTv2ygFy86HFU= +github.com/gin-gonic/gin v1.10.0/go.mod h1:4PMNQiOhvDRa013RKVbsiNwoyezlm2rm0uX/T7kzp5Y= +github.com/go-chi/chi/v5 v5.3.1 h1:3j4HZLGZQ3JpMCrPJF/Jl3mYJfWLKBfNJ6quurUGCf8= +github.com/go-chi/chi/v5 v5.3.1/go.mod h1:R+tYY2hNuVUUjxoPtqUdgBqevM9s9njzkTLutVsOCto= +github.com/go-darwin/apfs v0.0.0-20211011131704-f84b94dbf348 h1:JnrjqG5iR07/8k7NqrLNilRsl3s1EPRQEGvbPyOce68= +github.com/go-darwin/apfs v0.0.0-20211011131704-f84b94dbf348/go.mod h1:Czxo/d1g948LtrALAZdL04TL/HnkopquAjxYUuI02bo= +github.com/go-faster/errors v0.8.0 h1:9T9eJrM+72dFk7n4DfhuaDDe6cyuFCSW2oNUkN77Yqc= +github.com/go-faster/errors v0.8.0/go.mod h1:5ySTjWFiphBs07IKuiL69nxdfd5+fzh1u7FPGZP2quo= +github.com/go-faster/jx v1.2.0 h1:T2YHJPrFaYu21fJtUxC9GzmluKu8rVIFDwwGBKTDseI= +github.com/go-faster/jx v1.2.0/go.mod h1:UWLOVDmMG597a5tBFPLIWJdUxz5/2emOpfsj9Neg0PE= +github.com/go-faster/xor v1.0.0 h1:2o8vTOgErSGHP3/7XwA5ib1FTtUsNtwCoLLBjl31X38= +github.com/go-faster/xor v1.0.0/go.mod h1:x5CaDY9UKErKzqfRfFZdfu+OSTfoZny3w5Ak7UxcipQ= +github.com/go-faster/yaml v0.4.6 h1:lOK/EhI04gCpPgPhgt0bChS6bvw7G3WwI8xxVe0sw9I= +github.com/go-faster/yaml v0.4.6/go.mod h1:390dRIvV4zbnO7qC9FGo6YYutc+wyyUSHBgbXL52eXk= +github.com/go-git/go-billy/v5 v5.9.0 h1:jItGXszUDRtR/AlferWPTMN4j38BQ88XnXKbilmmBPA= +github.com/go-git/go-billy/v5 v5.9.0/go.mod h1:jCnQMLj9eUgGU7+ludSTYoZL/GGmii14RxKFj7ROgHw= +github.com/go-logr/logr v1.2.2/go.mod h1:jdQByPbusPIv2/zmleS9BjJVeZ6kBagPoEUsqbVz/1A= +github.com/go-logr/logr v1.4.3 h1:CjnDlHq8ikf6E492q6eKboGOC0T8CDaOvkHCIg8idEI= +github.com/go-logr/logr v1.4.3/go.mod h1:9T104GzyrTigFIr8wt5mBrctHMim0Nb2HLGrmQ40KvY= +github.com/go-logr/stdr v1.2.2 h1:hSWxHoqTgW2S2qGc0LTAI563KZ5YKYRhT3MFKZMbjag= +github.com/go-logr/stdr v1.2.2/go.mod h1:mMo/vtBO5dYbehREoey6XUKy/eSumjCCveDpRre4VKE= +github.com/go-ole/go-ole v1.2.6/go.mod h1:pprOEPIfldk/42T2oK7lQ4v4JSDwmV0As9GaiUsvbm0= +github.com/go-ole/go-ole v1.3.0 h1:Dt6ye7+vXGIKZ7Xtk4s6/xVdGDQynvom7xCFEdWr6uE= +github.com/go-ole/go-ole v1.3.0/go.mod h1:5LS6F96DhAwUc7C+1HLexzMXY1xGRSryjyPPKW6zv78= +github.com/go-openapi/errors v0.22.8 h1:oP7sW7TWc3wFFjrzzj0nI83H2qMBkNjNfSd+XRejk/I= +github.com/go-openapi/errors v0.22.8/go.mod h1:BuUoHcYrU6E7V9gfj1I5wLQqgtIHnup/alXZ8KdgQ0w= +github.com/go-openapi/strfmt v0.27.0 h1:kbcTeaD9TXuXD0hhMXzuYa1sdTo6+dWGvwjW93E80IM= +github.com/go-openapi/strfmt v0.27.0/go.mod h1:s/qhDqfY72irigXUGJmtgid2Rm+3tnz3k8hZaRmvWYc= +github.com/go-openapi/testify/v2 v2.6.0 h1:5PKH2HE7YJ/LuRPQGvSxBRlFXNQhSetBLlGAgUEu3ug= +github.com/go-openapi/testify/v2 v2.6.0/go.mod h1:SgsVHtfooshd0tublTtJ50FPKhujf47YRqauXXOUxfw= +github.com/go-playground/assert/v2 v2.2.0 h1:JvknZsQTYeFEAhQwI4qEt9cyV5ONwRHC+lYKSsYSR8s= +github.com/go-playground/assert/v2 v2.2.0/go.mod h1:VDjEfimB/XKnb+ZQfWdccd7VUvScMdVu0Titje2rxJ4= +github.com/go-playground/locales v0.14.1 h1:EWaQ/wswjilfKLTECiXz7Rh+3BjFhfDFKv/oXslEjJA= +github.com/go-playground/locales v0.14.1/go.mod h1:hxrqLVvrK65+Rwrd5Fc6F2O76J/NuW9t0sjnWqG1slY= +github.com/go-playground/universal-translator v0.18.1 h1:Bcnm0ZwsGyWbCzImXv+pAJnYK9S473LQFuzCbDbfSFY= +github.com/go-playground/universal-translator v0.18.1/go.mod h1:xekY+UJKNuX9WP91TpwSH2VMlDf28Uj24BCp08ZFTUY= +github.com/go-playground/validator/v10 v10.30.3 h1:4MU6YkEwx7GbcPJOZxrtbu+QfF3pJLJuaYTeAH0DYy8= +github.com/go-playground/validator/v10 v10.30.3/go.mod h1:4Axh7oCNGcoGkqLoE4YWt6n20mcEIsPRlB7vPk3lpyc= +github.com/go-resty/resty/v2 v2.17.2 h1:FQW5oHYcIlkCNrMD2lloGScxcHJ0gkjshV3qcQAyHQk= +github.com/go-resty/resty/v2 v2.17.2/go.mod h1:kCKZ3wWmwJaNc7S29BRtUhJwy7iqmn+2mLtQrOyQlVA= +github.com/go-task/slim-sprig/v3 v3.0.0 h1:sUs3vkvUymDpBKi3qH1YSqBQk9+9D/8M2mN1vB6EwHI= +github.com/go-task/slim-sprig/v3 v3.0.0/go.mod h1:W848ghGpv3Qj3dhTPRyJypKRiqCdHZiAzKg9hl15HA8= +github.com/go-viper/mapstructure/v2 v2.5.0 h1:vM5IJoUAy3d7zRSVtIwQgBj7BiWtMPfmPEgAXnvj1Ro= +github.com/go-viper/mapstructure/v2 v2.5.0/go.mod h1:oJDH3BJKyqBA2TXFhDsKDGDTlndYOZ6rGS0BRZIxGhM= +github.com/goccy/go-json v0.10.6 h1:p8HrPJzOakx/mn/bQtjgNjdTcN+/S6FcG2CTtQOrHVU= +github.com/goccy/go-json v0.10.6/go.mod h1:oq7eo15ShAhp70Anwd5lgX2pLfOS3QCiwU/PULtXL6M= +github.com/gofrs/flock v0.13.0 h1:95JolYOvGMqeH31+FC7D2+uULf6mG61mEZ/A8dRYMzw= +github.com/gofrs/flock v0.13.0/go.mod h1:jxeyy9R1auM5S6JYDBhDt+E2TCo7DkratH4Pgi8P+Z0= +github.com/gogo/protobuf v1.3.2 h1:Ov1cvc58UF3b5XjBnZv7+opcTcQFZebYjWzi34vdm4Q= +github.com/gogo/protobuf v1.3.2/go.mod h1:P1XiOD3dCwIKUDQYPy72D8LYyHL2YPYrpS2s69NZV8Q= +github.com/golang-jwt/jwt/v4 v4.5.2 h1:YtQM7lnr8iZ+j5q71MGKkNw9Mn7AjHM68uc9g5fXeUI= +github.com/golang-jwt/jwt/v4 v4.5.2/go.mod h1:m21LjoU+eqJr34lmDMbreY2eSTRJ1cv77w39/MY0Ch0= +github.com/golang-jwt/jwt/v5 v5.3.1 h1:kYf81DTWFe7t+1VvL7eS+jKFVWaUnK9cB1qbwn63YCY= +github.com/golang-jwt/jwt/v5 v5.3.1/go.mod h1:fxCRLWMO43lRc8nhHWY6LGqRcf+1gQWArsqaEUEa5bE= +github.com/golang/protobuf v1.5.4 h1:i7eJL8qZTpSEXOPTxNKhASYpMn+8e5Q6AdndVa1dWek= +github.com/golang/protobuf v1.5.4/go.mod h1:lnTiLA8Wa4RWRcIUkrtSVa5nRhsEGBg48fD6rSs7xps= +github.com/google/btree v1.1.3 h1:CVpQJjYgC4VbzxeGVHfvZrv1ctoYCAI8vbl07Fcxlyg= +github.com/google/btree v1.1.3/go.mod h1:qOPhT0dTNdNzV6Z/lhRX0YXUafgPLFUh+gZMl761Gm4= +github.com/google/flatbuffers v25.12.19+incompatible h1:haMV2JRRJCe1998HeW/p0X9UaMTK6SDo0ffLn2+DbLs= +github.com/google/flatbuffers v25.12.19+incompatible/go.mod h1:1AeVuKshWv4vARoZatz6mlQ0JxURH0Kv5+zNeJKJCa8= +github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8= +github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU= +github.com/google/pprof v0.0.0-20240509144519-723abb6459b7 h1:velgFPYr1X9TDwLIfkV7fWqsFlf7TeP11M/7kPd/dVI= +github.com/google/pprof v0.0.0-20240509144519-723abb6459b7/go.mod h1:kf6iHlnVGwgKolg33glAes7Yg/8iWP8ukqeldJSO7jw= +github.com/google/s2a-go v0.1.9 h1:LGD7gtMgezd8a/Xak7mEWL0PjoTQFvpRudN895yqKW0= +github.com/google/s2a-go v0.1.9/go.mod h1:YA0Ei2ZQL3acow2O62kdp9UlnvMmU7kA6Eutn0dXayM= +github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0= +github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo= +github.com/googleapis/enterprise-certificate-proxy v0.3.18 h1:hvVi34VucdrV1IIsiWuqYM8kutw/92MxNEFxCJZEh0k= +github.com/googleapis/enterprise-certificate-proxy v0.3.18/go.mod h1:rSEsBUemEBZEexP2y6jPp16LUmUbjmSbcPMQizR0o4k= +github.com/googleapis/gax-go/v2 v2.22.0 h1:PjIWBpgGIVKGoCXuiCoP64altEJCj3/Ei+kSU5vlZD4= +github.com/googleapis/gax-go/v2 v2.22.0/go.mod h1:irWBbALSr0Sk3qlqb9SyJ1h68WjgeFuiOzI4Rqw5+aY= +github.com/gopherjs/gopherjs v0.0.0-20181017120253-0766667cb4d1 h1:EGx4pi6eqNxGaHF6qqu48+N2wcFQ5qg5FXgOdqsJ5d8= +github.com/gopherjs/gopherjs v0.0.0-20181017120253-0766667cb4d1/go.mod h1:wJfORRmW1u3UXTncJ5qlYoELFm8eSnnEO6hX4iZ3EWY= +github.com/gorilla/schema v1.4.1 h1:jUg5hUjCSDZpNGLuXQOgIWGdlgrIdYvgQ0wZtdK1M3E= +github.com/gorilla/schema v1.4.1/go.mod h1:Dg5SSm5PV60mhF2NFaTV1xuYYj8tV8NOPRo4FggUMnM= +github.com/gorilla/securecookie v1.1.1 h1:miw7JPhV+b/lAHSXz4qd/nN9jRiAFV5FwjeKyCS8BvQ= +github.com/gorilla/securecookie v1.1.1/go.mod h1:ra0sb63/xPlUeL+yeDciTfxMRAA+MP+HVt/4epWDjd4= +github.com/gorilla/sessions v1.2.1 h1:DHd3rPN5lE3Ts3D8rKkQ8x/0kqfeNmBAaiSi+o7FsgI= +github.com/gorilla/sessions v1.2.1/go.mod h1:dk2InVEVJ0sfLlnXv9EAgkf6ecYs/i80K/zI+bUmuGM= +github.com/gotd/contrib v0.20.0 h1:1Wc4+HMQiIKYQuGHVwVksIx152HFTP6B5n88dDe0ZYw= +github.com/gotd/contrib v0.20.0/go.mod h1:P6o8W4niqhDPHLA0U+SA/L7l3BQHYLULpeHfRSePn9o= +github.com/gotd/ige v0.3.0 h1:4f6LEHWsVDLBG0bT9wWG2/9TZb5aWm265G8ZlTXmRRU= +github.com/gotd/ige v0.3.0/go.mod h1:FE9bTaQtvfArizAcZuI4sS6gXaEUBmixdUufVHoCKac= +github.com/gotd/log v0.1.0 h1:4LJUEvafD1xtBwx2QkrlzFnRgbYXTlWqJPDi8BvrLbU= +github.com/gotd/log v0.1.0/go.mod h1:5ilhdu1Ux0QvDY/FF3Ojfw24Ws3SlCtyLwOpXy8KYXs= +github.com/gotd/log/logzap v0.1.1 h1:O6l7d8HUbODe+UMcrM47eXYDwdJ6RNmpQejLjrlcEIQ= +github.com/gotd/log/logzap v0.1.1/go.mod h1:5ObZkITbfhbsBOLzBkzmMk9QxXc0eNQpimau7zRL+Y8= +github.com/gotd/neo v0.1.5 h1:oj0iQfMbGClP8xI59x7fE/uHoTJD7NZH9oV1WNuPukQ= +github.com/gotd/neo v0.1.5/go.mod h1:9A2a4bn9zL6FADufBdt7tZt+WMhvZoc5gWXihOPoiBQ= +github.com/gotd/td v0.140.0 h1:trNBzTnhNtNwHsFp5qwKnNxQRAZJ6/BRE+uH3Lojauk= +github.com/gotd/td v0.140.0/go.mod h1:0ZkRxG7N+5ooG7/zdRXcnGautGPM6IKmyPQvdsAeF20= +github.com/gotd/td v0.161.0 h1:krbzsb70cakdrqF+MUIo+W7BkQTVhyB1kNS7X/+BLcY= +github.com/gotd/td v0.161.0/go.mod h1:7HdCs+zeJugdgZAF5iG8f70eOJvuiH2QzjoyUcysXbY= +github.com/hashicorp/errwrap v1.0.0/go.mod h1:YH+1FKiLXxHSkmPseP+kNlulaMuP3n2brvKWEqk/Jc4= +github.com/hashicorp/errwrap v1.1.0 h1:OxrOeh75EUXMY8TBjag2fzXGZ40LB6IKw45YeGUDY2I= +github.com/hashicorp/errwrap v1.1.0/go.mod h1:YH+1FKiLXxHSkmPseP+kNlulaMuP3n2brvKWEqk/Jc4= +github.com/hashicorp/go-cleanhttp v0.5.2 h1:035FKYIWjmULyFRBKPs8TBQoi0x6d9G4xc9neXJWAZQ= +github.com/hashicorp/go-cleanhttp v0.5.2/go.mod h1:kO/YDlP8L1346E6Sodw+PrpBSV4/SoxCXGY6BqNFT48= +github.com/hashicorp/go-hclog v1.6.3 h1:Qr2kF+eVWjTiYmU7Y31tYlP1h0q/X3Nl3tPGdaB11/k= +github.com/hashicorp/go-hclog v1.6.3/go.mod h1:W4Qnvbt70Wk/zYJryRzDRU/4r0kIg0PVHBcfoyhpF5M= +github.com/hashicorp/go-multierror v1.1.1 h1:H5DkEtf6CXdFp0N0Em5UCwQpXMWke8IA0+lD48awMYo= +github.com/hashicorp/go-multierror v1.1.1/go.mod h1:iw975J/qwKPdAO1clOe2L8331t/9/fmwbPZ6JB6eMoM= +github.com/hashicorp/go-retryablehttp v0.7.8 h1:ylXZWnqa7Lhqpk0L1P1LzDtGcCR0rPVUrx/c8Unxc48= +github.com/hashicorp/go-retryablehttp v0.7.8/go.mod h1:rjiScheydd+CxvumBsIrFKlx3iS0jrZ7LvzFGFmuKbw= +github.com/hashicorp/go-uuid v1.0.2/go.mod h1:6SBZvOh/SIDV7/2o3Jml5SYk/TvGqwFJ/bN7x4byOro= +github.com/hashicorp/go-uuid v1.0.3 h1:2gKiV6YVmrJ1i2CKKa9obLvRieoRGviZFL26PcT/Co8= +github.com/hashicorp/go-uuid v1.0.3/go.mod h1:6SBZvOh/SIDV7/2o3Jml5SYk/TvGqwFJ/bN7x4byOro= +github.com/inconshreveable/mousetrap v1.1.0 h1:wN+x4NVGpMsO7ErUn/mUI3vEoE6Jt13X2s0bqwp9tc8= +github.com/inconshreveable/mousetrap v1.1.0/go.mod h1:vpF70FUmC8bwa3OWnCshd2FqLfsEA9PFc4w1p2J65bw= +github.com/internxt/rclone-adapter v0.0.0-20260708165336-dd6561bacfa2 h1:ZeebPK9Bnpy40uTuneEeVMYaMMerol3qsmIvzD0EIAk= +github.com/internxt/rclone-adapter v0.0.0-20260708165336-dd6561bacfa2/go.mod h1:4jGLEnNHyWOVSGn89IeWUVqlKCEitOM3D32/XypGy/o= +github.com/iyear/connectproxy v0.1.1 h1:JZOF/62vvwRGBWcgSyWRb0BpKD4FSs0++B5/y5pNE4c= +github.com/iyear/connectproxy v0.1.1/go.mod h1:yD4zOmSMQCmwHIT4fk8mg4k2M15z8VoMSoeY6NNJdsA= +github.com/iyear/tdl/core v0.20.4 h1:3CuYn56XRdSvyZwJq9hi4Oi4nGyItIR0WiURdeZwsA0= +github.com/iyear/tdl/core v0.20.4/go.mod h1:8CdaYmd2ZOCzoUdXMW0cATfnAwA9RFtNNXZDwF+odMU= +github.com/jcmturner/aescts/v2 v2.0.0 h1:9YKLH6ey7H4eDBXW8khjYslgyqG2xZikXP0EQFKrle8= +github.com/jcmturner/aescts/v2 v2.0.0/go.mod h1:AiaICIRyfYg35RUkr8yESTqvSy7csK90qZ5xfvvsoNs= +github.com/jcmturner/dnsutils/v2 v2.0.0 h1:lltnkeZGL0wILNvrNiVCR6Ro5PGU/SeBvVO/8c/iPbo= +github.com/jcmturner/dnsutils/v2 v2.0.0/go.mod h1:b0TnjGOvI/n42bZa+hmXL+kFJZsFT7G4t3HTlQ184QM= +github.com/jcmturner/gofork v1.7.6 h1:QH0l3hzAU1tfT3rZCnW5zXl+orbkNMMRGJfdJjHVETg= +github.com/jcmturner/gofork v1.7.6/go.mod h1:1622LH6i/EZqLloHfE7IeZ0uEJwMSUyQ/nDd82IeqRo= +github.com/jcmturner/goidentity/v6 v6.0.1 h1:VKnZd2oEIMorCTsFBnJWbExfNN7yZr3EhJAxwOkZg6o= +github.com/jcmturner/goidentity/v6 v6.0.1/go.mod h1:X1YW3bgtvwAXju7V3LCIMpY0Gbxyjn/mY9zx4tFonSg= +github.com/jcmturner/gokrb5/v8 v8.4.4 h1:x1Sv4HaTpepFkXbt2IkL29DXRf8sOfZXo8eRKh687T8= +github.com/jcmturner/gokrb5/v8 v8.4.4/go.mod h1:1btQEpgT6k+unzCwX1KdWMEwPPkkgBtP+F6aCACiMrs= +github.com/jcmturner/rpc/v2 v2.0.3 h1:7FXXj8Ti1IaVFpSAziCZWNzbNuZmnvw/i6CqLNdWfZY= +github.com/jcmturner/rpc/v2 v2.0.3/go.mod h1:VUJYCIDm3PVOEHw8sgt091/20OJjskO/YJki3ELg/Hc= +github.com/jlaffaye/ftp v0.2.1-0.20251026020404-6602e981a1bb h1:6vkM8gO+zFV2m21QzGYyUSq5TP0VQgP2Xz3UQyCN2kI= +github.com/jlaffaye/ftp v0.2.1-0.20251026020404-6602e981a1bb/go.mod h1:H1+whwD0Qe3YOunlXIWhh3rlvzW5cZfkMDYGQPg+KAM= +github.com/json-iterator/go v1.1.12 h1:PV8peI4a0ysnczrg+LtxykD8LfKY9ML6u2jnxaEnrnM= +github.com/json-iterator/go v1.1.12/go.mod h1:e30LSqwooZae/UwlEbR2852Gd8hjQvJoHmT4TnhNGBo= +github.com/jtolds/gls v4.20.0+incompatible h1:xdiiI2gbIgH/gLH7ADydsJ1uDOEzR8yvV7C0MuV77Wo= +github.com/jtolds/gls v4.20.0+incompatible/go.mod h1:QJZ7F/aHp+rZTRtaJ1ow/lLfFfVYBRgL+9YlvaHOwJU= +github.com/jtolio/noiseconn v0.0.0-20231127013910-f6d9ecbf1de7 h1:JcltaO1HXM5S2KYOYcKgAV7slU0xPy1OcvrVgn98sRQ= +github.com/jtolio/noiseconn v0.0.0-20231127013910-f6d9ecbf1de7/go.mod h1:MEkhEPFwP3yudWO0lj6vfYpLIB+3eIcuIW+e0AZzUQk= +github.com/jzelinskie/whirlpool v0.0.0-20201016144138-0675e54bb004 h1:G+9t9cEtnC9jFiTxyptEKuNIAbiN5ZCQzX2a74lj3xg= +github.com/jzelinskie/whirlpool v0.0.0-20201016144138-0675e54bb004/go.mod h1:KmHnJWQrgEvbuy0vcvj00gtMqbvNn1L+3YUZLK/B92c= +github.com/keybase/go-keychain v0.0.1 h1:way+bWYa6lDppZoZcgMbYsvC7GxljxrskdNInRtuthU= +github.com/keybase/go-keychain v0.0.1/go.mod h1:PdEILRW3i9D8JcdM+FmY6RwkHGnhHxXwkPPMeUgOK1k= +github.com/kisielk/errcheck v1.5.0/go.mod h1:pFxgyoBC7bSaBwPgfKdkLd5X25qrDl4LWUI2bnpBCr8= +github.com/kisielk/gotool v1.0.0/go.mod h1:XhKaO+MFFWcvkIS/tQcRk01m1F5IRFswLeQ+oQHNcck= +github.com/klauspost/compress v1.19.2 h1:hMRETovs/pu/dVWN7zIT1PGG8t509MwT6bO7XSi26R8= +github.com/klauspost/compress v1.19.2/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ= +github.com/klauspost/cpuid/v2 v2.4.0 h1:S6Hrbc7+ywsr0r+RLapfGBHfyefhCTwEh3A0tV913Dw= +github.com/klauspost/cpuid/v2 v2.4.0/go.mod h1:19jmZ9mjzoF//ddRSUsv0zfBTJWh3QJh9FNxZTMrGxU= +github.com/koofr/go-httpclient v0.0.0-20240520111329-e20f8f203988 h1:CjEMN21Xkr9+zwPmZPaJJw+apzVbjGL5uK/6g9Q2jGU= +github.com/koofr/go-httpclient v0.0.0-20240520111329-e20f8f203988/go.mod h1:/agobYum3uo/8V6yPVnq+R82pyVGCeuWW5arT4Txn8A= +github.com/koofr/go-koofrclient v0.0.0-20221207135200-cbd7fc9ad6a6 h1:FHVoZMOVRA+6/y4yRlbiR3WvsrOcKBd/f64H7YiWR2U= +github.com/koofr/go-koofrclient v0.0.0-20221207135200-cbd7fc9ad6a6/go.mod h1:MRAz4Gsxd+OzrZ0owwrUHc0zLESL+1Y5syqK/sJxK2A= +github.com/kr/fs v0.1.0 h1:Jskdu9ieNAYnjxsi0LbQp1ulIKZV1LAFgK1tWhpZgl8= +github.com/kr/fs v0.1.0/go.mod h1:FFnZGqtBN9Gxj7eW1uZ42v5BccTP0vu6NEaFoC2HwRg= +github.com/kr/pretty v0.2.1/go.mod h1:ipq/a2n7PKx3OHsz4KJII5eveXtPO4qwEXGdVfWzfnI= +github.com/kr/pretty v0.3.1 h1:flRD4NNwYAUpkphVc1HcthR4KEIFJ65n8Mw5qdRn3LE= +github.com/kr/pretty v0.3.1/go.mod h1:hoEshYVHaxMs3cyo3Yncou5ZscifuDolrwPKZanG3xk= +github.com/kr/pty v1.1.1/go.mod h1:pFQYn66WHrOpPYNljwOMqo10TkYh1fy3cYio2l3bCsQ= +github.com/kr/text v0.1.0/go.mod h1:4Jbv+DJW3UT/LiOwJeYQe1efqtUx/iVham/4vfdArNI= +github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY= +github.com/kr/text v0.2.0/go.mod h1:eLer722TekiGuMkidMxC/pM04lWEeraHUUmBw8l2grE= +github.com/kylelemons/godebug v1.1.0 h1:RPNrshWIDI6G2gRW9EHilWtl7Z6Sb1BR0xunSBf0SNc= +github.com/kylelemons/godebug v1.1.0/go.mod h1:9/0rRGxNHcop5bhtWyNeEfOS8JIWk580+fNqagV/RAw= +github.com/lanrat/extsort v1.4.2 h1:akbLIdo4PhNZtvjpaWnbXtGMmLtnGzXplkzfgl+XTTY= +github.com/lanrat/extsort v1.4.2/go.mod h1:hceP6kxKPKebjN1RVrDBXMXXECbaI41Y94tt6MDazc4= +github.com/leodido/go-urn v1.4.0 h1:WT9HwE9SGECu3lg4d/dIA+jxlljEa1/ffXKmRjqdmIQ= +github.com/leodido/go-urn v1.4.0/go.mod h1:bvxc+MVxLKB4z00jd1z+Dvzr47oO32F/QSNjSBOlFxI= +github.com/lpar/calendar v0.2.0 h1:A1kxv6sbvBHFUkd2XotanIRqEXQGreQOeuGhkJqIaRA= +github.com/lpar/calendar v0.2.0/go.mod h1:fsVJa4o2NvXYzaCE5RXMadqbZzWEhNCGOUPVHTIZdjc= +github.com/lufia/plan9stats v0.0.0-20260627054121-477a66015f15 h1:YkjVPl/YH5XlJ+/NiwzJtPYXXKRcyjmEUhsDci6YK3c= +github.com/lufia/plan9stats v0.0.0-20260627054121-477a66015f15/go.mod h1:autxFIvghDt3jPTLoqZ9OZ7s9qTGNAWmYCjVFWPX/zg= +github.com/mailru/easyjson v0.9.2 h1:dX8U45hQsZpxd80nLvDGihsQ/OxlvTkVUXH2r/8cb2M= +github.com/mailru/easyjson v0.9.2/go.mod h1:1+xMtQp2MRNVL/V1bOzuP3aP8VNwRW55fQUto+XFtTU= +github.com/mattn/go-colorable v0.1.15 h1:+u9SLTRGnXv73cEsnsmoZBom+dMU88B2M0aDcWy0/jY= +github.com/mattn/go-colorable v0.1.15/go.mod h1:6LmQG8QLFO4G5z1gPvYEzlUgJ2wF+stgPZH1UqBm1s8= +github.com/mattn/go-isatty v0.0.23 h1:cYwCQTQf3HB6xUC+BtyCLZNr7IzbOmoZbmssVNzSyiQ= +github.com/mattn/go-isatty v0.0.23/go.mod h1:nMCL3Zebbrt45jsMDgnfIwz6ydEQApk5oEI3HqDio6A= +github.com/mattn/go-runewidth v0.0.3/go.mod h1:LwmH8dsx7+W8Uxz3IHJYH5QSwggIsqBzpuz5H//U1FU= +github.com/mattn/go-runewidth v0.0.24 h1:cpokDiIn0MGnhdHwuWnJBITySJ20QyNGnY2kR/ay2DU= +github.com/mattn/go-runewidth v0.0.24/go.mod h1:XBkDxAl56ILZc9knddidhrOlY5R/pDhgLpndooCuJAs= +github.com/mitchellh/go-homedir v1.1.0 h1:lukF9ziXFxDFPkA1vsr5zpc1XuPDn/wFntq5mG+4E0Y= +github.com/mitchellh/go-homedir v1.1.0/go.mod h1:SfyaCUpYCn1Vlf4IUYiD9fPX4A5wJrkLzIz1N1q0pr0= +github.com/moby/sys/mountinfo v0.7.2 h1:1shs6aH5s4o5H2zQLn796ADW1wMrIwHsyJ2v9KouLrg= +github.com/moby/sys/mountinfo v0.7.2/go.mod h1:1YOa8w8Ih7uW0wALDUgT1dTTSBrZ+HiBLGws92L2RU4= +github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd h1:TRLaZ9cD/w8PVh93nsPXa1VrQ6jlwL5oN8l14QlcNfg= +github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd/go.mod h1:6dJC0mAP4ikYIbvyc7fijjWJddQyLn8Ig3JB5CqoB9Q= +github.com/modern-go/reflect2 v1.0.2 h1:xBagoLtFs94CBntxluKeaWgTMpvLxC4ur3nMaC9Gz0M= +github.com/modern-go/reflect2 v1.0.2/go.mod h1:yWuevngMOJpCy52FWWMvUC8ws7m/LJsjYzDa0/r8luk= +github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 h1:C3w9PqII01/Oq1c1nUAm88MOHcQC9l5mIlSMApZMrHA= +github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822/go.mod h1:+n7T8mK8HuQTcFwEeznm/DIxMOiR9yIdICNftLE1DvQ= +github.com/ncw/swift/v2 v2.0.5 h1:9o5Gsd7bInAFEqsGPcaUdsboMbqf8lnNtxqWKFT9iz8= +github.com/ncw/swift/v2 v2.0.5/go.mod h1:cbAO76/ZwcFrFlHdXPjaqWZ9R7Hdar7HpjRXBfbjigk= +github.com/nxadm/tail v1.4.8 h1:nPr65rt6Y5JFSKQO7qToXr7pePgD6Gwiw05lkbyAQTE= +github.com/nxadm/tail v1.4.8/go.mod h1:+ncqLTQzXmGhMZNUePPaPqPvBxHAIsmXswZKocGu+AU= +github.com/ogen-go/ogen v1.23.0 h1:QaWeKm2KZ2zy7NkqqO1Vdl5idNqlG+svxdgwVAX+zbo= +github.com/ogen-go/ogen v1.23.0/go.mod h1:bwwvC3AmCV+LrL5lazyQwwof90402mdcSyI0FOzzpfM= +github.com/oklog/ulid/v2 v2.1.1 h1:suPZ4ARWLOJLegGFiZZ1dFAkqzhMjL3J1TzI+5wHz8s= +github.com/oklog/ulid/v2 v2.1.1/go.mod h1:rcEKHmBBKfef9DhnvX7y1HZBYxjXb0cP5ExxNsTT1QQ= +github.com/onsi/ginkgo v1.16.5 h1:8xi0RTUf59SOSfEtZMvwTvXYMzG4gV23XVHOZiXNtnE= +github.com/onsi/ginkgo v1.16.5/go.mod h1:+E8gABHa3K6zRBolWtd+ROzc/U5bkGt0FwiG042wbpU= +github.com/onsi/ginkgo/v2 v2.17.3 h1:oJcvKpIb7/8uLpDDtnQuf18xVnwKp8DTD7DQ6gTd/MU= +github.com/onsi/ginkgo/v2 v2.17.3/go.mod h1:nP2DPOQoNsQmsVyv5rDA8JkXQoCs6goXIvr/PRJ1eCc= +github.com/onsi/gomega v1.37.0 h1:CdEG8g0S133B4OswTDC/5XPSzE1OeP29QOioj2PID2Y= +github.com/onsi/gomega v1.37.0/go.mod h1:8D9+Txp43QWKhM24yyOBEdpkzN8FvJyAwecBgsU4KU0= +github.com/oracle/oci-go-sdk/v65 v65.121.0 h1:1J+5ARgrodrx8kzFy/hxznaoUzz43jr0EestCzEaOHw= +github.com/oracle/oci-go-sdk/v65 v65.121.0/go.mod h1:Pzy+BpgkDesvGZXEHgslwhIYobHCPHg6wRta1mWnlqQ= +github.com/panjf2000/ants/v2 v2.12.1 h1:BWvU2wHpyXWxhhNXsGB6JXLCNbshyLd1QxvoAmZnu10= +github.com/panjf2000/ants/v2 v2.12.1/go.mod h1:tSQuaNQ6r6NRhPt+IZVUevvDyFMTs+eS4ztZc52uJTY= +github.com/patrickmn/go-cache v2.1.0+incompatible h1:HRMgzkcYKYpi3C8ajMPV8OFXaaRUnok+kx1WdO15EQc= +github.com/patrickmn/go-cache v2.1.0+incompatible/go.mod h1:3Qf8kWWT7OJRJbdiICTKqZju1ZixQ/KpMGzzAfe6+WQ= +github.com/pborman/getopt v0.0.0-20170112200414-7148bc3a4c30/go.mod h1:85jBQOZwpVEaDAr341tbn15RS4fCAsIst0qp7i8ex1o= +github.com/pelletier/go-toml/v2 v2.2.4 h1:mye9XuhQ6gvn5h28+VilKrrPoQVanw5PMw/TB0t5Ec4= +github.com/pelletier/go-toml/v2 v2.2.4/go.mod h1:2gIqNv+qfxSVS7cM2xJQKtLSTLUE9V8t9Stt+h56mCY= +github.com/pengsrc/go-shared v0.2.1-0.20190131101655-1999055a4a14 h1:XeOYlK9W1uCmhjJSsY78Mcuh7MVkNjTzmHx1yBzizSU= +github.com/pengsrc/go-shared v0.2.1-0.20190131101655-1999055a4a14/go.mod h1:jVblp62SafmidSkvWrXyxAme3gaTfEtWwRPGz5cpvHg= +github.com/peterh/liner v1.2.2 h1:aJ4AOodmL+JxOZZEL2u9iJf8omNRpqHc/EbrK+3mAXw= +github.com/peterh/liner v1.2.2/go.mod h1:xFwJyiKIXJZUKItq5dGHZSTBRAuG/CpeNpWLyiNRNwI= +github.com/pierrec/lz4/v4 v4.1.27 h1:+PhzhWDrjRj89TH2sw43nE3+4+W8lSxIuQadEHZyjUk= +github.com/pierrec/lz4/v4 v4.1.27/go.mod h1:EoQMVJgeeEOMsCqCzqFm2O0cJvljX2nGZjcRIPL34O4= +github.com/pkg/browser v0.0.0-20240102092130-5ac0b6a4141c h1:+mdjkGKdHQG3305AYmdv1U2eRNDiU2ErMBj1gwrq8eQ= +github.com/pkg/browser v0.0.0-20240102092130-5ac0b6a4141c/go.mod h1:7rwL4CYBLnjLxUqIJNnCWiEdr3bn6IUYi15bNlnbCCU= +github.com/pkg/diff v0.0.0-20200914180035-5b29258ca4f7/go.mod h1:zO8QMzTeZd5cpnIkz/Gn6iK0jDfGicM1nynOkkPIl28= +github.com/pkg/errors v0.9.1 h1:FEBLx1zS214owpjy7qsBeixbURkuhQAwrK5UwLGTwt4= +github.com/pkg/errors v0.9.1/go.mod h1:bwawxfHBFNV+L2hUp1rHADufV3IMtnDRdf1r5NINEl0= +github.com/pkg/sftp v1.13.11 h1:0N92SLTB8JqASJB14ZLHHzFnBV8mG9zw4K7jghEFWuE= +github.com/pkg/sftp v1.13.11/go.mod h1:uNkH9roSXglNJqM+glJJi+TQXQUm0fXFWqCFmT8hsN0= +github.com/pkg/xattr v0.4.12 h1:rRTkSyFNTRElv6pkA3zpjHpQ90p/OdHQC1GmGh1aTjM= +github.com/pkg/xattr v0.4.12/go.mod h1:di8WF84zAKk8jzR1UBTEWh9AUlIZZ7M/JNt8e9B6ktU= +github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4= +github.com/power-devops/perfstat v0.0.0-20240221224432-82ca36839d55 h1:o4JXh1EVt9k/+g42oCprj/FisM4qX9L3sZB3upGN2ZU= +github.com/power-devops/perfstat v0.0.0-20240221224432-82ca36839d55/go.mod h1:OmDBASR4679mdNQnz2pUhc2G8CO2JrUAVFDRBDP/hJE= +github.com/pquerna/otp v1.5.0 h1:NMMR+WrmaqXU4EzdGJEE1aUUI0AMRzsp96fFFWNPwxs= +github.com/pquerna/otp v1.5.0/go.mod h1:dkJfzwRKNiegxyNb54X/3fLwhCynbMspSyWKnvi1AEg= +github.com/prometheus/client_golang v1.23.2 h1:Je96obch5RDVy3FDMndoUsjAhG5Edi49h0RJWRi/o0o= +github.com/prometheus/client_golang v1.23.2/go.mod h1:Tb1a6LWHB3/SPIzCoaDXI4I8UHKeFTEQ1YCr+0Gyqmg= +github.com/prometheus/client_model v0.6.2 h1:oBsgwpGs7iVziMvrGhE53c/GrLUsZdHnqNwqPLxwZyk= +github.com/prometheus/client_model v0.6.2/go.mod h1:y3m2F6Gdpfy6Ut/GBsUqTWZqCUvMVzSfMLjcu6wAwpE= +github.com/prometheus/common v0.70.0 h1:bcpru3tWPVnxGnETLgOV5jbp/JRXgYEyv65CuBLAMMI= +github.com/prometheus/common v0.70.0/go.mod h1:S/SFasQmgGiYH6C81LKCtYa8QACgthGg5zxL2udV7SY= +github.com/prometheus/procfs v0.21.1 h1:GljZCt+zSTS+NZq88cyQ1LjZ+RCHp3uVuabBWA5+OJI= +github.com/prometheus/procfs v0.21.1/go.mod h1:aB55Cww9pdSJVHk0hUf0inxWyyjPogFIjmHKYgMKmtY= +github.com/putdotio/go-putio/putio v0.0.0-20200123120452-16d982cac2b8 h1:Y258uzXU/potCYnQd1r6wlAnoMB68BiCkCcCnKx1SH8= +github.com/putdotio/go-putio/putio v0.0.0-20200123120452-16d982cac2b8/go.mod h1:bSJjRokAHHOhA+XFxplld8w2R/dXLH7Z3BZ532vhFwU= +github.com/quic-go/quic-go v0.59.0 h1:OLJkp1Mlm/aS7dpKgTc6cnpynnD2Xg7C1pwL6vy/SAw= +github.com/quic-go/quic-go v0.59.0/go.mod h1:upnsH4Ju1YkqpLXC305eW3yDZ4NfnNbmQRCMWS58IKU= +github.com/rclone/Proton-API-Bridge v1.0.5 h1:K1++Qtk3PvgkiCCiv6Pahju1TMOzKY6VSwiwT7XLAVc= +github.com/rclone/Proton-API-Bridge v1.0.5/go.mod h1:vCeOPhlXzevN0AFojgh1zsjhetiShy/ArvJ/xkFUDWk= +github.com/rclone/go-proton-api v1.0.4 h1:AJW0e9pB4j0hVK4WqyGErFwaI+5MUQWPCtj5FYYxtPg= +github.com/rclone/go-proton-api v1.0.4/go.mod h1:QAlkFfswzrBuxvCORWV8rZdddg52hahMN98CFWoFW1E= +github.com/rclone/rclone v1.75.1 h1:kIxQcoDLj2Gke/gMSHK7OnxhX1Gu1cJBLP1kJZoaFp0= +github.com/rclone/rclone v1.75.1/go.mod h1:4zmMjGatCkSJPRZDpo+7y3xOl8S29EMUyKvZop5mHr4= +github.com/refraction-networking/utls v1.8.2 h1:j4Q1gJj0xngdeH+Ox/qND11aEfhpgoEvV+S9iJ2IdQo= +github.com/refraction-networking/utls v1.8.2/go.mod h1:jkSOEkLqn+S/jtpEHPOsVv/4V4EVnelwbMQl4vCWXAM= +github.com/relvacode/iso8601 v1.7.0 h1:BXy+V60stMP6cpswc+a93Mq3e65PfXCgDFfhvNNGrdo= +github.com/relvacode/iso8601 v1.7.0/go.mod h1:FlNp+jz+TXpyRqgmM7tnzHHzBnz776kmAH2h3sZCn0I= +github.com/rfjakob/eme v1.2.0 h1:8dAHL+WVAw06+7DkRKnRiFp1JL3QjcJEZFqDnndUaSI= +github.com/rfjakob/eme v1.2.0/go.mod h1:cVvpasglm/G3ngEfcfT/Wt0GwhkuO32pf/poW6Nyk1k= +github.com/rogpeppe/go-internal v1.15.0 h1:D0RCU5rMAp+SpgkiNdrjfJ+LX4J1M32V2NeCY7EJ6hc= +github.com/rogpeppe/go-internal v1.15.0/go.mod h1:DrUVZyrJU+txYW5/1kwtXQSMFio52ZOxX7yM1VHvnxs= +github.com/sabhiram/go-gitignore v0.0.0-20210923224102-525f6e181f06 h1:OkMGxebDjyw0ULyrTYWeN0UNCCkmCWfjPnIA2W6oviI= +github.com/sabhiram/go-gitignore v0.0.0-20210923224102-525f6e181f06/go.mod h1:+ePHsJ1keEjQtpvf9HHw0f4ZeJ0TLRsxhunSI2hYJSs= +github.com/samber/lo v1.53.0 h1:t975lj2py4kJPQ6haz1QMgtId2gtmfktACxIXArw3HM= +github.com/samber/lo v1.53.0/go.mod h1:4+MXEGsJzbKGaUEQFKBq2xtfuznW9oz/WrgyzMzRoM0= +github.com/segmentio/asm v1.2.1 h1:DTNbBqs57ioxAD4PrArqftgypG4/qNpXoJx8TVXxPR0= +github.com/segmentio/asm v1.2.1/go.mod h1:BqMnlJP91P8d+4ibuonYZw9mfnzI9HfxselHZr5aAcs= +github.com/sergi/go-diff v1.0.0/go.mod h1:0CfEIISq7TuYL3j771MWULgwwjU+GofnZX9QAmXWZgo= +github.com/shirou/gopsutil/v4 v4.26.6 h1:Mzr/npDtQC/xpeEuQKHZt8Zo9CmPvhTj8nkR8w5TLDs= +github.com/shirou/gopsutil/v4 v4.26.6/go.mod h1:LZ6ewCSkBqUpvSOf+LsTGnRinC6iaNUNMGBtDkJBaLQ= +github.com/shopspring/decimal v1.4.0 h1:bxl37RwXBklmTi0C79JfXCEBD1cqqHt0bbgBAGFp81k= +github.com/shopspring/decimal v1.4.0/go.mod h1:gawqmDU56v4yIKSwfBSFip1HdCCXN8/+DMd9qYNcwME= +github.com/sirupsen/logrus v1.7.0/go.mod h1:yWOB1SBYBC5VeMP7gHvWumXLIWorT60ONWic61uBYv0= +github.com/sirupsen/logrus v1.9.4 h1:TsZE7l11zFCLZnZ+teH4Umoq5BhEIfIzfRDZ1Uzql2w= +github.com/sirupsen/logrus v1.9.4/go.mod h1:ftWc9WdOfJ0a92nsE2jF5u5ZwH8Bv2zdeOC42RjbV2g= +github.com/skratchdot/open-golang v0.0.0-20200116055534-eef842397966 h1:JIAuq3EEf9cgbU6AtGPK4CTG3Zf6CKMNqf0MHTggAUA= +github.com/skratchdot/open-golang v0.0.0-20200116055534-eef842397966/go.mod h1:sUM3LWHvSMaG192sy56D9F7CNvL7jUJVXoqM1QKLnog= +github.com/smarty/assertions v1.16.0 h1:EvHNkdRA4QHMrn75NZSoUQ/mAUXAYWfatfB01yTCzfY= +github.com/smarty/assertions v1.16.0/go.mod h1:duaaFdCS0K9dnoM50iyek/eYINOZ64gbh1Xlf6LG7AI= +github.com/smartystreets/goconvey v1.8.1 h1:qGjIddxOk4grTu9JPOU31tVfq3cNdBlNa5sSznIX1xY= +github.com/smartystreets/goconvey v1.8.1/go.mod h1:+/u4qLyY6x1jReYOp7GOM2FSt8aP9CzCZL03bI28W60= +github.com/snabb/httpreaderat v1.0.1 h1:whlb+vuZmyjqVop8x1EKOg05l2NE4z9lsMMXjmSUCnY= +github.com/snabb/httpreaderat v1.0.1/go.mod h1:lpbGrKDWF37yvRbtRvQsbesS6Ty5c83t8ztannPoMsA= +github.com/sony/gobreaker/v2 v2.4.0 h1:g2KJRW1Ubty3+ZOcSEUN7K+REQJdN6yo6XvaML+jptg= +github.com/sony/gobreaker/v2 v2.4.0/go.mod h1:pTyFJgcZ3h2tdQVLZZruK2C0eoFL1fb/G83wK1ZQl+s= +github.com/spacemonkeygo/monkit/v3 v3.0.25-0.20251022131615-eb24eb109368 h1:GyYC5Ntqk/yy9lEIGE7chdIvt4zP44taycwd9YDSGdc= +github.com/spacemonkeygo/monkit/v3 v3.0.25-0.20251022131615-eb24eb109368/go.mod h1:XkZYGzknZwkD0AKUnZaSXhRiVTLCkq7CWVa3IsE72gA= +github.com/spf13/cobra v1.10.2 h1:DMTTonx5m65Ic0GOoRY2c16WCbHxOOw6xxezuLaBpcU= +github.com/spf13/cobra v1.10.2/go.mod h1:7C1pvHqHw5A4vrJfjNwvOdzYu0Gml16OCs2GRiTUUS4= +github.com/spf13/pflag v1.0.10 h1:4EBh2KAYBwaONj6b2Ye1GiHfwjqyROoF4RwYO+vPwFk= +github.com/spf13/pflag v1.0.10/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg= +github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+wExME= +github.com/stretchr/objx v0.4.0/go.mod h1:YvHI0jy2hoMjB+UWwv71VJQ9isScKT/TqJzVSSt89Yw= +github.com/stretchr/objx v0.5.0/go.mod h1:Yh+to48EsGEfYuaHDzXPcE3xhTkx73EhmCGUpEOglKo= +github.com/stretchr/objx v0.5.3 h1:jmXUvGomnU1o3W/V5h2VEradbpJDwGrzugQQvL0POH4= +github.com/stretchr/objx v0.5.3/go.mod h1:rDQraq+vQZU7Fde9LOZLr8Tax6zZvy4kuNKF+QYS+U0= +github.com/stretchr/testify v1.2.2/go.mod h1:a8OnRcib4nhh0OaRAV+Yts87kKdq0PP7pXfy6kDkUVs= +github.com/stretchr/testify v1.3.0/go.mod h1:M5WIy9Dh21IEIfnGCwXGc5bZfKNJtfHm1UVUgZn+9EI= +github.com/stretchr/testify v1.3.1-0.20190311161405-34c6fa2dc709/go.mod h1:M5WIy9Dh21IEIfnGCwXGc5bZfKNJtfHm1UVUgZn+9EI= +github.com/stretchr/testify v1.4.0/go.mod h1:j7eGeouHqKxXV5pUuKE4zz7dFj8WfuZ+81PSLYec5m4= +github.com/stretchr/testify v1.6.1/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg= +github.com/stretchr/testify v1.7.0/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg= +github.com/stretchr/testify v1.7.1/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg= +github.com/stretchr/testify v1.8.0/go.mod h1:yNjHg4UonilssWZ8iaSj1OCr/vHnekPRkoO+kdMU+MU= +github.com/stretchr/testify v1.8.1/go.mod h1:w2LPCIKwWwSfY2zedu0+kehJoqGctiVI29o6fzry7u4= +github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWDWE= +github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg= +github.com/t3rm1n4l/go-mega v0.0.0-20260717075258-c6acd6a5bd04 h1:s30A8dMuZ55lUOi5xUTh1hlfLqHpVBsRMfpk2iKCayk= +github.com/t3rm1n4l/go-mega v0.0.0-20260717075258-c6acd6a5bd04/go.mod h1:BF/l2jNyK+2h/BJZ7VLMAz6m/IWjA2F67gTjV1C/+Bo= +github.com/tailscale/depaware v0.0.0-20210622194025-720c4b409502/go.mod h1:p9lPsd+cx33L3H9nNoecRRxPssFKUwwI50I3pZ0yT+8= +github.com/tklauser/go-sysconf v0.4.0 h1:7H0uAN+7RkwWRaxhYXDLqa5V3LPrJeV8wmD9dRUgPQU= +github.com/tklauser/go-sysconf v0.4.0/go.mod h1:8mTNWyog7H+MpKijp4VmKJAd2bbYQ2zuUwkYRbUArPI= +github.com/tklauser/numcpus v0.12.0 h1:NR85qdvHA9pFse3x3weVZ0r0ST8R6l5RHbZrlRaqob4= +github.com/tklauser/numcpus v0.12.0/go.mod h1:ABHeXzJnr/qqwguhClkZKT1/8VABcYrsyUiUGobwWJg= +github.com/twitchyliquid64/golang-asm v0.15.1 h1:SU5vSMR7hnwNxj24w34ZyCi/FmDZTkS4MhqMhdFk5YI= +github.com/twitchyliquid64/golang-asm v0.15.1/go.mod h1:a1lVb/DtPvCB8fslRZhAngC2+aY1QWCk3Cedj/Gdt08= +github.com/tyler-smith/go-bip39 v1.1.0 h1:5eUemwrMargf3BSLRRCalXT93Ns6pQJIjYQN2nyfOP8= +github.com/tyler-smith/go-bip39 v1.1.0/go.mod h1:gUYDtqQw1JS3ZJ8UWVcGTGqqr6YIN3CWg+kkNaLt55U= +github.com/ugorji/go/codec v1.2.12 h1:9LC83zGrHhuUA9l16C9AHXAqEV/2wBQ4nkvumAE65EE= +github.com/ugorji/go/codec v1.2.12/go.mod h1:UNopzCgEMSXjBc6AOMqYvWC1ktqTAfzJZUZgYf6w6lg= +github.com/ulikunitz/xz v0.5.15 h1:9DNdB5s+SgV3bQ2ApL10xRc35ck0DuIX/isZvIk+ubY= +github.com/ulikunitz/xz v0.5.15/go.mod h1:nbz6k7qbPmH4IRqmfOplQw/tblSgqTqBwxkY0oWt/14= +github.com/unknwon/goconfig v1.0.0 h1:rS7O+CmUdli1T+oDm7fYj1MwqNWtEJfNj+FqcUHML8U= +github.com/unknwon/goconfig v1.0.0/go.mod h1:qu2ZQ/wcC/if2u32263HTVC39PeOQRSmidQk3DuDFQ8= +github.com/wk8/go-ordered-map/v2 v2.1.8 h1:5h/BUHu93oj4gIdvHHHGsScSTMijfx5PeYkE/fJgbpc= +github.com/wk8/go-ordered-map/v2 v2.1.8/go.mod h1:5nJHM5DyteebpVlHnWMV0rPz6Zp7+xBAnxjb1X5vnTw= +github.com/xanzy/ssh-agent v0.3.3 h1:+/15pJfg/RsTxqYcX6fHqOXZwwMP+2VyYWJeWM2qQFM= +github.com/xanzy/ssh-agent v0.3.3/go.mod h1:6dzNDKs0J9rVPHPhaGCukekBHKqfl+L3KghI1Bc68Uw= +github.com/xyproto/randomstring v1.0.5 h1:YtlWPoRdgMu3NZtP45drfy1GKoojuR7hmRcnhZqKjWU= +github.com/xyproto/randomstring v1.0.5/go.mod h1:rgmS5DeNXLivK7YprL0pY+lTuhNQW3iGxZ18UQApw/E= +github.com/youmark/pkcs8 v0.0.0-20240726163527-a2c0da244d78 h1:ilQV1hzziu+LLM3zUTJ0trRztfwgjqKnBWNtSRkbmwM= +github.com/youmark/pkcs8 v0.0.0-20240726163527-a2c0da244d78/go.mod h1:aL8wCCfTfSfmXjznFBSZNN13rSJjlIOI1fUNAtF7rmI= +github.com/yuin/goldmark v1.1.27/go.mod h1:3hX8gzYuyVAZsxl0MRgGTJEmQBFcNTphYh9decYSb74= +github.com/yuin/goldmark v1.2.1/go.mod h1:3hX8gzYuyVAZsxl0MRgGTJEmQBFcNTphYh9decYSb74= +github.com/yuin/goldmark v1.4.13/go.mod h1:6yULJ656Px+3vBD8DxQVa3kxgyrAnzto9xy5taEt/CY= +github.com/yuin/goldmark v1.8.4 h1:oat/nd3U6NeQqFEL3xpEJq7d7c86NI+DbSNGAs4xnjA= +github.com/yuin/goldmark v1.8.4/go.mod h1:ip/1k0VRfGynBgxOz0yCqHrbZXhcjxyuS66Brc7iBKg= +github.com/yunify/qingstor-sdk-go/v3 v3.2.0 h1:9sB2WZMgjwSUNZhrgvaNGazVltoFUUfuS9f0uCWtTr8= +github.com/yunify/qingstor-sdk-go/v3 v3.2.0/go.mod h1:KciFNuMu6F4WLk9nGwwK69sCGKLCdd9f97ac/wfumS4= +github.com/yusufpapurcu/wmi v1.2.4 h1:zFUKzehAFReQwLys1b/iSMl+JQGSCSjtVqQn9bBrPo0= +github.com/yusufpapurcu/wmi v1.2.4/go.mod h1:SBZ9tNy3G9/m5Oi98Zks0QjeHVDvuK0qfxQmPyzfmi0= +github.com/zeebo/assert v1.3.1 h1:vukIABvugfNMZMQO1ABsyQDJDTVQbn+LWSMy1ol1h6A= +github.com/zeebo/assert v1.3.1/go.mod h1:Pq9JiuJQpG8JLJdtkwrJESF0Foym2/D9XMU5ciN/wJ0= +github.com/zeebo/blake3 v0.2.4 h1:KYQPkhpRtcqh0ssGYcKLG1JYvddkEA8QwCM/yBqhaZI= +github.com/zeebo/blake3 v0.2.4/go.mod h1:7eeQ6d2iXWRGF6npfaxl2CU+xy2Fjo2gxeyZGCRUjcE= +github.com/zeebo/errs v1.4.0 h1:XNdoD/RRMKP7HD0UhJnIzUy74ISdGGxURlYG8HSWSfM= +github.com/zeebo/errs v1.4.0/go.mod h1:sgbWHsvVuTPHcqJJGQ1WhI5KbWlHYz+2+2C/LSEtCw4= +github.com/zeebo/mwc v0.0.7 h1:0NerGhCww6ZQx+/xCx5iwznftveokvto1KILpYfENZk= +github.com/zeebo/mwc v0.0.7/go.mod h1:0B32or6moOig1YGuqMoimBpU9QK9uYaGG2bBOuddqtE= +github.com/zeebo/pcg v1.0.1 h1:lyqfGeWiv4ahac6ttHs+I5hwtH/+1mrhlCtVNQM2kHo= +github.com/zeebo/pcg v1.0.1/go.mod h1:09F0S9iiKrwn9rlI5yjLkmrug154/YRW6KnnXVDM/l4= +github.com/zeebo/xxh3 v1.1.0 h1:s7DLGDK45Dyfg7++yxI0khrfwq9661w9EN78eP/UZVs= +github.com/zeebo/xxh3 v1.1.0/go.mod h1:IisAie1LELR4xhVinxWS5+zf1lA4p0MW4T+w+W07F5s= +go.etcd.io/bbolt v1.5.0 h1:S7GAl7Fxv12yohbwFfIbQCGDWbQbtDGPET4P/bD4lxU= +go.etcd.io/bbolt v1.5.0/go.mod h1:mkltfYE5aUHQxUct9N9V+Kp7aSjFqjgrhcXIS70Lrdk= +go.opentelemetry.io/auto/sdk v1.2.1 h1:jXsnJ4Lmnqd11kwkBV2LgLoFMZKizbCi5fNZ/ipaZ64= +go.opentelemetry.io/auto/sdk v1.2.1/go.mod h1:KRTj+aOaElaLi+wW1kO/DZRXwkF4C5xPbEe3ZiIhN7Y= +go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.69.0 h1:8tvICD4vSTOOsNrsI4Ljf6C+6UKvpTEH5XY3JMoyPoo= +go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.69.0/go.mod h1:z9+yiacE0IHRqM4qFfkbt/JYlmYXgss8GY/jXoNuPJI= +go.opentelemetry.io/otel v1.44.0 h1:JjwHmHpA4iZ3wBxluu2fbbE7j4kqlE8jXyAyPXH7HqU= +go.opentelemetry.io/otel v1.44.0/go.mod h1:BMgjTHL9WPRlRjL2oZCBTL4whCGtXch2H4BhOPIAyYc= +go.opentelemetry.io/otel/metric v1.44.0 h1:1w0gILTcHdr3YI+ixLyjemwrVnsMURbTZFrSYCdDdmc= +go.opentelemetry.io/otel/metric v1.44.0/go.mod h1:8O7hanEPBNgEMmybD3s2VBKcgWOCsA6tzHBPODAiquo= +go.opentelemetry.io/otel/sdk v1.44.0 h1:nHYwb9lK+fJPU/dnT6s7W7Z8itMWyqrnVfbheVYrZ58= +go.opentelemetry.io/otel/sdk v1.44.0/go.mod h1:Osuydd3Se74nqjAKxid74N5eC+jfEqfTegHRnq58oK0= +go.opentelemetry.io/otel/sdk/metric v1.44.0 h1:3LlKgI+VjbVsjNRFZJZAJ30WjXC5VkNRks6si09iEfI= +go.opentelemetry.io/otel/sdk/metric v1.44.0/go.mod h1:5B5pMARnXxKhltooO4xUuCBorl65a4EpnTalObqOigA= +go.opentelemetry.io/otel/trace v1.44.0 h1:jxF5CsGYCe74MCRx2X4g7WsY/VBKRqqpNvXlX/6gtIk= +go.opentelemetry.io/otel/trace v1.44.0/go.mod h1:oLl1jrMQAVo6v3GAggN+1VH9VIz9iUSvW53sW1Q8PIE= +go.uber.org/atomic v1.11.0 h1:ZvwS0R+56ePWxUNi+Atn9dWONBPp/AUETXlHW0DxSjE= +go.uber.org/atomic v1.11.0/go.mod h1:LUxbIzbOniOlMKjJjyPfpl4v+PKK2cNJn91OQbhoJI0= +go.uber.org/goleak v1.3.0 h1:2K3zAYmnTNqV73imy9J1T3WC+gmCePx2hEGkimedGto= +go.uber.org/goleak v1.3.0/go.mod h1:CoHD4mav9JJNrW/WLlf7HGZPjdw8EucARQHekz1X6bE= +go.uber.org/multierr v1.11.0 h1:blXXJkSxSSfBVBlC76pxqeO+LN3aDfLQo+309xJstO0= +go.uber.org/multierr v1.11.0/go.mod h1:20+QtiLqy0Nd6FdQB9TLXag12DsQkrbs3htMFfDN80Y= +go.uber.org/zap v1.28.0 h1:IZzaP1Fv73/T/pBMLk4VutPl36uNC+OSUh3JLG3FIjo= +go.uber.org/zap v1.28.0/go.mod h1:rDLpOi171uODNm/mxFcuYWxDsqWSAVkFdX4XojSKg/Q= +go.yaml.in/yaml/v2 v2.4.4 h1:tuyd0P+2Ont/d6e2rl3be67goVK4R6deVxCUX5vyPaQ= +go.yaml.in/yaml/v2 v2.4.4/go.mod h1:gMZqIpDtDqOfM0uNfy0SkpRhvUryYH0Z6wdMYcacYXQ= +go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw= +go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg= +golang.org/x/arch v0.14.0 h1:z9JUEZWr8x4rR0OU6c4/4t6E6jOZ8/QBS2bBYBm4tx4= +golang.org/x/arch v0.14.0/go.mod h1:FEVrYAQjsQXMVJ1nsMoVVXPZg6p2JE2mx8psSWTDQys= +golang.org/x/crypto v0.0.0-20190308221718-c2843e01d9a2/go.mod h1:djNgcEr1/C05ACkg1iLfiJU5Ep61QUkGW8qpdssI0+w= +golang.org/x/crypto v0.0.0-20191011191535-87dc89f01550/go.mod h1:yigFU9vqHzYiE8UmvKecakEJjdnWj3jj499lnFckfCI= +golang.org/x/crypto v0.0.0-20200622213623-75b288015ac9/go.mod h1:LzIPMQfyMNhhGPhUkYOs5KpL4U8rLKemX1yGLhDgUto= +golang.org/x/crypto v0.0.0-20210322153248-0c34fe9e7dc2/go.mod h1:T9bdIzuCu7OtxOm1hfPfRQxPLYneinmdGuTeoZ9dtd4= +golang.org/x/crypto v0.0.0-20210921155107-089bfa567519/go.mod h1:GvvjBRRGRdwPK5ydBHafDWAxML/pGHZbMvKqRZ5+Abc= +golang.org/x/crypto v0.0.0-20220622213112-05595931fe9d/go.mod h1:IxCIyHEi3zRg3s0A5j5BB6A9Jmi73HwBIUl50j+osU4= +golang.org/x/crypto v0.4.0/go.mod h1:3quD/ATkf6oY+rnes5c3ExXTbLc8mueNue5/DoinL80= +golang.org/x/crypto v0.6.0/go.mod h1:OFC/31mSvZgRz0V1QTNCzfAI1aIRzbiufJtkMIlEp58= +golang.org/x/crypto v0.7.0/go.mod h1:pYwdfH91IfpZVANVyUOhSIPZaFoJGxTFbZhFTx+dXZU= +golang.org/x/crypto v0.56.0 h1:GUh5Ii4J5jtcseSMiRqr1jXCNHoxjeV9Fmekc2oLy6Y= +golang.org/x/crypto v0.56.0/go.mod h1:OMW5y6CY9l38uPLmxU6l6pwcXp1obtLo3e6gT7gQR2I= +golang.org/x/exp v0.0.0-20260709172345-9ea1abe57597 h1:qLvzZeaANDgyVOA8pyHCOStGlXn0rseXma+GQjeuv2g= +golang.org/x/exp v0.0.0-20260709172345-9ea1abe57597/go.mod h1:EdfpwwqSu+0Li0mzskwHU6FWDV3t9Q+RZDo3QMUtL3Q= +golang.org/x/image v0.45.0 h1:FMb1nTbH5H9vF55SriQHgFw5GnNL9Jg6L25BwXKzhB0= +golang.org/x/image v0.45.0/go.mod h1:n62x/7RqlwXDvGsSU4u6IUTUf6KghUZ9Bt7cG/T9Fx4= +golang.org/x/mod v0.2.0/go.mod h1:s0Qsj1ACt9ePp/hMypM3fl4fZqREWJwdYDEqhRiZZUA= +golang.org/x/mod v0.3.0/go.mod h1:s0Qsj1ACt9ePp/hMypM3fl4fZqREWJwdYDEqhRiZZUA= +golang.org/x/mod v0.4.0/go.mod h1:s0Qsj1ACt9ePp/hMypM3fl4fZqREWJwdYDEqhRiZZUA= +golang.org/x/mod v0.6.0-dev.0.20220419223038-86c51ed26bb4/go.mod h1:jJ57K6gSWd91VN4djpZkiMVwK6gcyfeH4XE8wZrZaV4= +golang.org/x/mod v0.8.0/go.mod h1:iBbtSCu2XBx23ZKBPSOrRkjjQPZFPuis4dIYUhu/chs= +golang.org/x/mod v0.38.0 h1:MECBjubtXD7yj4HrhIUcywNaGeNVUdfVnxmPajOk4yk= +golang.org/x/mod v0.38.0/go.mod h1:V6Xz0pq8TQ3dGqVQ1FVHuelZpAL0uNhSkk9ogYP3c40= +golang.org/x/net v0.0.0-20190404232315-eb5bcb51f2a3/go.mod h1:t9HGtf8HONx5eT2rtn7q6eTqICYqUVnKs3thJo3Qplg= +golang.org/x/net v0.0.0-20190620200207-3b0461eec859/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s= +golang.org/x/net v0.0.0-20200114155413-6afb5195e5aa/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s= +golang.org/x/net v0.0.0-20200226121028-0de0cce0169b/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s= +golang.org/x/net v0.0.0-20201021035429-f5854403a974/go.mod h1:sp8m0HH+o8qH0wwXwYZr8TS3Oi6o0r6Gce1SSxlDquU= +golang.org/x/net v0.0.0-20210226172049-e18ecbb05110/go.mod h1:m0MpNAwzfU5UDzcl9v0D8zg8gWTRqZa9RBIspLL5mdg= +golang.org/x/net v0.0.0-20211112202133-69e39bad7dc2/go.mod h1:9nx3DQGgdP8bBQD5qxJ1jj9UTztislL4KSBs9R2vV5Y= +golang.org/x/net v0.0.0-20220722155237-a158d28d115b/go.mod h1:XRhObCWvk6IyKnWLug+ECip1KBveYUHfp+8e9klMJ9c= +golang.org/x/net v0.3.0/go.mod h1:MBQ8lrhLObU/6UmLb4fmbmk5OcyYmqtbGd/9yIeKjEE= +golang.org/x/net v0.6.0/go.mod h1:2Tu9+aMcznHK/AK1HMvgo6xiTLG5rD5rZLDS+rp2Bjs= +golang.org/x/net v0.7.0/go.mod h1:2Tu9+aMcznHK/AK1HMvgo6xiTLG5rD5rZLDS+rp2Bjs= +golang.org/x/net v0.8.0/go.mod h1:QVkue5JL9kW//ek3r6jTKnTFis1tRmNAW2P1shuFdJc= +golang.org/x/net v0.58.0 h1:ynWG7rqYi4ccpTEuPZ2QGWHktVEM9DMCj9yzDE0Q7To= +golang.org/x/net v0.58.0/go.mod h1:YwCddHnFlT7eLQqVprV19OnhLGtc5xOKgE0RyqgfWAU= +golang.org/x/oauth2 v0.36.0 h1:peZ/1z27fi9hUOFCAZaHyrpWG5lwe0RJEEEeH0ThlIs= +golang.org/x/oauth2 v0.36.0/go.mod h1:YDBUJMTkDnJS+A4BP4eZBjCqtokkg1hODuPjwiGPO7Q= +golang.org/x/sync v0.0.0-20190423024810-112230192c58/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= +golang.org/x/sync v0.0.0-20190911185100-cd5d95a43a6e/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= +golang.org/x/sync v0.0.0-20201020160332-67f06af15bc9/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= +golang.org/x/sync v0.0.0-20201207232520-09787c993a3a/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= +golang.org/x/sync v0.0.0-20220722155255-886fb9371eb4/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= +golang.org/x/sync v0.1.0/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= +golang.org/x/sync v0.22.0 h1:SZjpbeLmrCk4xhRSZFNZW5gFUeCeFgjekvI/+gfScek= +golang.org/x/sync v0.22.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0= +golang.org/x/sys v0.0.0-20190215142949-d0b11bdaac8a/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY= +golang.org/x/sys v0.0.0-20190412213103-97732733099d/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20190916202348-b4ddaad3f8a3/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20191026070338-33540a1f6037/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20200930185726-fdedc70b468f/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20201119102817-f84b799fce68/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20201204225414-ed752295db88/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20210124154548-22da62e12c0c/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20210423082822-04245dca01da/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20210514084401-e8d321eab015/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.0.0-20210615035016-665e8c7367d1/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.0.0-20211007075335-d3039528d8ac/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.0.0-20211117180635-dee7805ff2e1/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.0.0-20220408201424-a24fb2fb8a0f/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.0.0-20220520151302-bc2c85ada10a/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.0.0-20220715151400-c0bba94af5f8/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.0.0-20220722155257-8c9f86f7a55f/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.1.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.3.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.5.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.6.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs= +golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw= +golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo= +golang.org/x/term v0.0.0-20210927222741-03fcf44c2211/go.mod h1:jbD1KX2456YbFQfuXm/mYQcufACuNUgVhRMnK/tPxf8= +golang.org/x/term v0.3.0/go.mod h1:q750SLmJuPmVoN1blW3UFBPREJfb1KmY3vwxfr+nFDA= +golang.org/x/term v0.5.0/go.mod h1:jMB1sMXY+tzblOD4FWmEbocvup2/aLOaQEp7JmGp78k= +golang.org/x/term v0.6.0/go.mod h1:m6U89DPEgQRMq3DNkDClhWw02AUbt2daBVO4cn4Hv9U= +golang.org/x/term v0.45.0 h1:NwWyBmoJCbfTHpxrWoZ9C6/VxOf7ic219I8xZZFdrf0= +golang.org/x/term v0.45.0/go.mod h1:9aqxs0blBcrm/n0L9QW0aRVD+ktan8ssZromtqJC43w= +golang.org/x/text v0.3.0/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ= +golang.org/x/text v0.3.3/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ= +golang.org/x/text v0.3.6/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ= +golang.org/x/text v0.3.7/go.mod h1:u+2+/6zg+i71rQMx5EYifcz6MCKuco9NR6JIITiCfzQ= +golang.org/x/text v0.5.0/go.mod h1:mrYo+phRRbMaCq/xk9113O4dZlRixOauAjOtrjsXDZ8= +golang.org/x/text v0.7.0/go.mod h1:mrYo+phRRbMaCq/xk9113O4dZlRixOauAjOtrjsXDZ8= +golang.org/x/text v0.8.0/go.mod h1:e1OnstbJyHTd6l/uOt8jFFHp6TRDWZR/bV3emEE/zU8= +golang.org/x/text v0.14.0/go.mod h1:18ZOQIKpY8NJVqYksKHtTdi31H5itFRjB5/qKTNYzSU= +golang.org/x/text v0.41.0 h1:vz/seA0lnX87Othu2f/0L24RcgrXD9/YFTSuGjj3rH8= +golang.org/x/text v0.41.0/go.mod h1:jvf1O8ajNzZqhSrQBPbutR/EB83Cc0CFrezNQIwbb5M= +golang.org/x/time v0.15.0 h1:bbrp8t3bGUeFOx08pvsMYRTCVSMk89u4tKbNOZbp88U= +golang.org/x/time v0.15.0/go.mod h1:Y4YMaQmXwGQZoFaVFk4YpCt4FLQMYKZe9oeV/f4MSno= +golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ= +golang.org/x/tools v0.0.0-20191119224855-298f0cb1881e/go.mod h1:b+2E5dAYhXwXZwtnZ6UAqBI28+e2cm9otk0dWdXHAEo= +golang.org/x/tools v0.0.0-20200619180055-7c47624df98f/go.mod h1:EkVYQZoAsY45+roYkvgYkIh4xh/qjgUK9TdY2XT94GE= +golang.org/x/tools v0.0.0-20201211185031-d93e913c1a58/go.mod h1:emZCQorbCU4vsT4fOWvOPXz4eW1wZW4PmDk9uLelYpA= +golang.org/x/tools v0.0.0-20210106214847-113979e3529a/go.mod h1:emZCQorbCU4vsT4fOWvOPXz4eW1wZW4PmDk9uLelYpA= +golang.org/x/tools v0.1.12/go.mod h1:hNGJHUnrk76NpqgfD5Aqm5Crs+Hm0VOH/i9J2+nxYbc= +golang.org/x/tools v0.6.0/go.mod h1:Xwgl3UAJ/d3gWutnCtw505GrjyAbvKui8lOU390QaIU= +golang.org/x/tools v0.48.0 h1:3+hClM1aLL5mjMKm5ovokw9epgRXPuu2tILgismM6RE= +golang.org/x/tools v0.48.0/go.mod h1:08xX0orndb/F7jJxGDicx061tyd5pcMto75YMAXr6lk= +golang.org/x/xerrors v0.0.0-20190717185122-a985d3407aa7/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0= +golang.org/x/xerrors v0.0.0-20191011141410-1b5146add898/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0= +golang.org/x/xerrors v0.0.0-20191204190536-9bdfabe68543/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0= +golang.org/x/xerrors v0.0.0-20200804184101-5ec99f83aff1/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0= +gonum.org/v1/gonum v0.17.0 h1:VbpOemQlsSMrYmn7T2OUvQ4dqxQXU+ouZFQsZOx50z4= +gonum.org/v1/gonum v0.17.0/go.mod h1:El3tOrEuMpv2UdMrbNlKEh9vd86bmQ6vqIcDwxEOc1E= +google.golang.org/api v0.279.0 h1:hsx2M2OaRcaKtVYK6vXEUnQvdjnend7ZYES+lYaot74= +google.golang.org/api v0.279.0/go.mod h1:B9TqLBwJqVjp1mtt7WeoQwWRwvu/400y5lETOql+giQ= +google.golang.org/genproto v0.0.0-20260319201613-d00831a3d3e7 h1:XzmzkmB14QhVhgnawEVsOn6OFsnpyxNPRY9QV01dNB0= +google.golang.org/genproto v0.0.0-20260319201613-d00831a3d3e7/go.mod h1:L43LFes82YgSonw6iTXTxXUX1OlULt4AQtkik4ULL/I= +google.golang.org/genproto/googleapis/api v0.0.0-20260706201446-f0a921348800 h1:admdQBe8jR3VWhBsUrAOaF2Qw6K/+p5pSm1GN8+6Fw4= +google.golang.org/genproto/googleapis/api v0.0.0-20260706201446-f0a921348800/go.mod h1:FPk7EXUKMtImne7AmknoYjT4QXqKIzzRbeQIXzLk6fQ= +google.golang.org/genproto/googleapis/rpc v0.0.0-20260715232425-e75dac1f907d h1:Jkpk39hlTZOIp3RbfvNX9R8Hv+Sw0X89nlU/xFOErsc= +google.golang.org/genproto/googleapis/rpc v0.0.0-20260715232425-e75dac1f907d/go.mod h1:4Hqkh8ycfw05ld/3BWL7rJOSfebL2Q+DVDeRgYgxUU8= +google.golang.org/grpc v1.84.0-dev.0.20260723093437-b6eac429d7b6 h1:HfjjkdGIa8u9sP9EW5WCygy0kQDuTI/Tax4j//t24Fo= +google.golang.org/grpc v1.84.0-dev.0.20260723093437-b6eac429d7b6/go.mod h1:ljCht0DrxQrXBDRTZp52Qxh3Ffk8CdYm2sj4O2QN2C0= +google.golang.org/protobuf v1.36.11 h1:fV6ZwhNocDyBLK0dj+fg8ektcVegBBuEolpbTQyBNVE= +google.golang.org/protobuf v1.36.11/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco= +gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= +gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c h1:Hei/4ADfdWqJk1ZMxUNpqntNwaWcugrBjAiHlqqRiVk= +gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c/go.mod h1:JHkPIbrfpd72SG/EVd6muEfDQjcINNoR0C8j2r3qZ4Q= +gopkg.in/natefinch/lumberjack.v2 v2.2.1 h1:bBRl1b0OH9s/DuPhuXpNl+VtCaJXFZ5/uEFST95x9zc= +gopkg.in/natefinch/lumberjack.v2 v2.2.1/go.mod h1:YD8tP3GAjkrDg1eZH7EGmyESg/lsYskCTPBJVb9jqSc= +gopkg.in/tomb.v1 v1.0.0-20141024135613-dd632973f1e7 h1:uRGJdciOHaEIrze2W8Q3AKkepLTh2hOroT7a+7czfdQ= +gopkg.in/tomb.v1 v1.0.0-20141024135613-dd632973f1e7/go.mod h1:dt/ZhP58zS4L8KSrWDmTeBkI65Dw0HsyUHuEVlX15mw= +gopkg.in/validator.v2 v2.0.1 h1:xF0KWyGWXm/LM2G1TrEjqOu4pa6coO9AlWSf3msVfDY= +gopkg.in/validator.v2 v2.0.1/go.mod h1:lIUZBlB3Im4s/eYp39Ry/wkR02yOPhZ9IwIRBjuPuG8= +gopkg.in/yaml.v2 v2.2.2/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI= +gopkg.in/yaml.v2 v2.4.0 h1:D8xgwECY7CYvx+Y2n4sBz93Jn9JRvxdiyyo8CTfuKaY= +gopkg.in/yaml.v2 v2.4.0/go.mod h1:RDklbk79AGWmwhnvt/jBztapEOGDOx6ZbXqjP6csGnQ= +gopkg.in/yaml.v3 v3.0.0-20200313102051-9f266ea9e77c/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= +gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA= +gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= +moul.io/http2curl/v2 v2.3.0 h1:9r3JfDzWPcbIklMOs2TnIFzDYvfAZvjeavG6EzP7jYs= +moul.io/http2curl/v2 v2.3.0/go.mod h1:RW4hyBjTWSYDOxapodpNEtX0g5Eb16sxklBqmd2RHcE= +nhooyr.io/websocket v1.8.17 h1:KEVeLJkUywCKVsnLIDlD/5gtayKp8VoCkksHCGGfT9Y= +nhooyr.io/websocket v1.8.17/go.mod h1:rN9OFWIUwuxg4fR5tELlYC04bXYowCP9GX47ivo2l+c= +rsc.io/qr v0.2.0 h1:6vBLea5/NRMVTz8V66gipeLycZMl/+UlFmk8DvqQ6WY= +rsc.io/qr v0.2.0/go.mod h1:IF+uZjkb9fqyeF/4tlBoynqmQxUoPfWEKh921coOuXs= +sigs.k8s.io/yaml v1.6.0 h1:G8fkbMSAFqgEFgh4b1wmtzDnioxFCUgTZhlbj5P9QYs= +sigs.k8s.io/yaml v1.6.0/go.mod h1:796bPqUfzR/0jLAl6XjHl3Ck7MiyVv8dbTdyT3/pMf4= +storj.io/common v0.0.0-20260629224719-ba1bff0a7846 h1:TRtVGWm/Y4KAi4coWMxVpydPvBQKYVhkKMar7wu9Zow= +storj.io/common v0.0.0-20260629224719-ba1bff0a7846/go.mod h1:1GZnCZNGbzBzaqhG0cUypQeLfNhNIeNY9JVMMiCF14M= +storj.io/drpc v1.0.0 h1:1Xf1KCXXbV1viIfN56eqdJ3cNwpAL7OKwQkpS6ksing= +storj.io/drpc v1.0.0/go.mod h1:Y9LZaa8esL1PW2IDMqJE7CFSNq7d5bQ3RI7mGPtmKMg= +storj.io/eventkit v0.0.0-20260716074419-6861a92e2aa5 h1:JPguPTktHaha9TBEHvuxZmB+mCkiMPgEMOLfosDoE9o= +storj.io/eventkit v0.0.0-20260716074419-6861a92e2aa5/go.mod h1:RdWvp249AGm9FZnVKP88y3hFa7vlx9Jf3Cy1P+4EXIU= +storj.io/infectious v0.0.2 h1:rGIdDC/6gNYAStsxsZU79D/MqFjNyJc1tsyyj9sTl7Q= +storj.io/infectious v0.0.2/go.mod h1:QEjKKww28Sjl1x8iDsjBpOM4r1Yp8RsowNcItsZJ1Vs= +storj.io/picobuf v0.0.4 h1:qswHDla+YZ2TovGtMnU4astjvrADSIz84FXRn0qgP6o= +storj.io/picobuf v0.0.4/go.mod h1:hSMxmZc58MS/2qSLy1I0idovlO7+6K47wIGUyRZa6mg= +storj.io/uplink v1.14.3 h1:b89nziD1JgiF6oya+kDSBsqWp9Dzp5ADCUdhA2e2zvs= +storj.io/uplink v1.14.3/go.mod h1:jAe47qR+gRnHW2+3THe4cWJBAj8ARZv4P9mguJRo7C4= diff --git a/internal/backends/all.go b/internal/backends/all.go new file mode 100644 index 0000000..90c2ec4 --- /dev/null +++ b/internal/backends/all.go @@ -0,0 +1,13 @@ +//go:build !slim + +// Package backends registers rclone storage backends by blank import. +// +// rclone resolves a remote by looking up its type in a registry that each +// backend populates from its own init(), so a backend that is not imported +// simply does not exist at runtime. The default build registers all of them, so +// any remote in the user's rclone.conf works and adding one later needs no +// rebuild. Cost is binary size: ~92 MB stripped, against ~50 MB for the slim +// build below. +package backends + +import _ "github.com/rclone/rclone/backend/all" diff --git a/internal/backends/slim.go b/internal/backends/slim.go new file mode 100644 index 0000000..d40903b --- /dev/null +++ b/internal/backends/slim.go @@ -0,0 +1,11 @@ +//go:build slim + +// Package backends registers rclone storage backends by blank import. +// +// The slim build registers only pikpak, the remote this tool was written +// against, trading generality for roughly half the binary size. A remote of any +// other type fails at resolution time with rclone's "didn't find backend" +// error, so build without the tag if the destination might change. +package backends + +import _ "github.com/rclone/rclone/backend/pikpak" diff --git a/internal/remote/fs.go b/internal/remote/fs.go new file mode 100644 index 0000000..6e71596 --- /dev/null +++ b/internal/remote/fs.go @@ -0,0 +1,131 @@ +// Package remote owns the rclone side: resolving a destination and reporting on it. +package remote + +import ( + "context" + "errors" + "fmt" + "os" + "strings" + "sync" + + "github.com/rclone/rclone/fs" + "github.com/rclone/rclone/fs/config" + "github.com/rclone/rclone/fs/config/configfile" +) + +// Tunables are rclone settings this tool overrides. +// +// The values exist because of pikpak: it commits an upload as a server-side +// async task, and rclone abandons a still-pending one once its low-level retries +// run out, failing a transfer that would have succeeded. Fewer parallel +// transfers keep that queue short; more retries wait it out. +type Tunables struct { + Transfers int + LowLevelRetries int +} + +// DefaultTunables are the values the shell pipeline settled on for pikpak. +func DefaultTunables() Tunables { + return Tunables{Transfers: 2, LowLevelRetries: 20} +} + +// installOnce guards configfile.Install, which swaps unsynchronised package +// globals in rclone's config package. Calling it twice is harmless on its own, +// but racing it against an Fs resolution is not. +var installOnce sync.Once + +// Init loads the user's rclone.conf and applies tunables to a derived context. +// +// Environment variables still win: rclone reads RCLONE_TRANSFERS and friends +// into its global config at package init, and fs.AddConfig copies that, so +// skipping the assignment when the variable is set preserves the operator's +// value. +// +// The config is loaded here, explicitly, because rclone's lazy path is fatal: +// config.LoadedData() calls fs.Fatalf on a config file it cannot parse or +// decrypt, and fs.Fatalf calls os.Exit(1) — past every defer, and with an exit +// code this tool defines as "incomplete", which would send a driver into an +// endless retry. Loading up front turns that into an ordinary error. +func Init(ctx context.Context, t Tunables) (context.Context, error) { + var err error + installOnce.Do(func() { + configfile.Install() + if lerr := config.Data().Load(); lerr != nil && !errors.Is(lerr, config.ErrorConfigFileNotFound) { + err = fmt.Errorf("cannot read rclone config %q: %w "+ + "(an encrypted config needs RCLONE_CONFIG_PASS)", config.GetConfigPath(), lerr) + } + }) + if err != nil { + return ctx, err + } + + ctx, ci := fs.AddConfig(ctx) + if !envSet("RCLONE_TRANSFERS") && t.Transfers > 0 { + ci.Transfers = t.Transfers + } + if !envSet("RCLONE_LOW_LEVEL_RETRIES") && t.LowLevelRetries > 0 { + ci.LowLevelRetries = t.LowLevelRetries + } + return ctx, nil +} + +// Resolve opens a destination given as an rclone REMOTE:PATH. +// +// rclone itself is the authority on what resolves: a remote can come from +// rclone.conf, from RCLONE_CONFIG__* environment variables with no config +// entry at all, from an inline `:type,opt=val:` connection string, or from a +// parameterised name like `pikpak,chunk_size=10M:path`. Pre-screening the name +// against the config sections would reject the last three, so the call is made +// first and the friendly "here is what you have configured" message is produced +// only for the one error that means the name was never defined. +func Resolve(ctx context.Context, remote string) (fs.Fs, error) { + if remote == "" { + return nil, fmt.Errorf("a destination remote is required (REMOTE:PATH)") + } + if !strings.Contains(remote, ":") { + return nil, fmt.Errorf("remote %q is not in rclone REMOTE:PATH form", remote) + } + + f, err := fs.NewFs(ctx, remote) + if err != nil { + if errors.Is(err, fs.ErrorNotFoundInConfigFile) { + return nil, fmt.Errorf("rclone remote %q is not configured; configured remotes: %s", + remote, strings.Join(sections(), ", ")) + } + return nil, fmt.Errorf("cannot reach %q — check credentials and connectivity: %w", remote, err) + } + return f, nil +} + +// FreeBytes reports free space on the remote. +// +// Backends without quota reporting return ok=false rather than an error: the +// shell pipeline treated an unanswerable quota as "unlimited" so a backend that +// cannot report never blocks a run, and that behaviour is preserved. +func FreeBytes(ctx context.Context, f fs.Fs) (free int64, ok bool) { + about := f.Features().About + if about == nil { + return 0, false + } + usage, err := about(ctx) + if err != nil || usage == nil || usage.Free == nil { + return 0, false + } + return *usage.Free, true +} + +// sections lists configured remote names. Safe only after Init has loaded the +// config without error. +func sections() []string { + out := config.FileSections() + if len(out) == 0 { + return []string{"(none)"} + } + return out +} + +func envSet(key string) bool { + _, ok := os.LookupEnv(key) + return ok +} diff --git a/internal/remote/fs_test.go b/internal/remote/fs_test.go new file mode 100644 index 0000000..3dbd30c --- /dev/null +++ b/internal/remote/fs_test.go @@ -0,0 +1,117 @@ +package remote + +import ( + "context" + "os" + "strings" + "testing" + + "github.com/rclone/rclone/fs" +) + +// Resolve's own validation runs before rclone is consulted, so these cases are +// checkable without a config file or a network. +func TestResolveRejectsMalformedDestinations(t *testing.T) { + tests := []struct { + name string + remote string + want string + }{ + {"empty", "", "destination remote is required"}, + {"no colon", "pikpak", "not in rclone REMOTE:PATH form"}, + {"path only", "/tmp/staging", "not in rclone REMOTE:PATH form"}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + _, err := Resolve(context.Background(), tt.remote) + if err == nil { + t.Fatalf("Resolve(%q) succeeded, want an error", tt.remote) + } + if !strings.Contains(err.Error(), tt.want) { + t.Errorf("Resolve(%q) error = %v, want it to mention %q", tt.remote, err, tt.want) + } + }) + } +} + +// Forms rclone accepts must not be rejected by our own pre-checks. An earlier +// version screened the name against the config file's sections, which turned +// away env-defined remotes, connection strings, and parameterised names that +// rclone resolves perfectly well. These must get past our validation and fail — +// if at all — inside rclone, on their own merits. +func TestResolveDefersUnusualFormsToRclone(t *testing.T) { + forms := []string{ + "pikpak,chunk_size=10M:mychannel", // parameterised remote name + ":local:/tmp", // inline connection string + "envonly:bucket", // possibly defined by RCLONE_CONFIG_ENVONLY_* + } + + for _, form := range forms { + t.Run(form, func(t *testing.T) { + _, err := Resolve(context.Background(), form) + if err == nil { + return // rclone resolved it; nothing to assert + } + if strings.Contains(err.Error(), "not in rclone REMOTE:PATH form") { + t.Errorf("Resolve(%q) was rejected by our own syntax check; "+ + "rclone should be the authority on what resolves: %v", form, err) + } + }) + } +} + +func TestDefaultTunablesMatchPikpakSettings(t *testing.T) { + // These two numbers are the outcome of debugging pikpak's async-commit + // behaviour in the shell pipeline; a silent change would reintroduce + // transfers that fail while the server is still committing. + got := DefaultTunables() + if got.Transfers != 2 { + t.Errorf("Transfers = %d, want 2", got.Transfers) + } + if got.LowLevelRetries != 20 { + t.Errorf("LowLevelRetries = %d, want 20", got.LowLevelRetries) + } +} + +func TestInitAppliesTunables(t *testing.T) { + // Unset both so the defaults, not an operator override, are what lands. + unset(t, "RCLONE_TRANSFERS") + unset(t, "RCLONE_LOW_LEVEL_RETRIES") + + ctx, err := Init(context.Background(), Tunables{Transfers: 2, LowLevelRetries: 20}) + if err != nil { + t.Fatalf("Init: %v", err) + } + ci := fs.GetConfig(ctx) + if ci.Transfers != 2 { + t.Errorf("ctx Transfers = %d, want 2", ci.Transfers) + } + if ci.LowLevelRetries != 20 { + t.Errorf("ctx LowLevelRetries = %d, want 20", ci.LowLevelRetries) + } +} + +// An operator's environment override must survive Init, so a tunable is only +// applied when the corresponding variable is absent. +func TestInitLeavesEnvOverridesAlone(t *testing.T) { + unset(t, "RCLONE_TRANSFERS") + t.Setenv("RCLONE_TRANSFERS", "7") + + ctx, err := Init(context.Background(), Tunables{Transfers: 2, LowLevelRetries: 20}) + if err != nil { + t.Fatalf("Init: %v", err) + } + if got := fs.GetConfig(ctx).Transfers; got == 2 { + t.Errorf("Transfers = 2; Init overwrote the RCLONE_TRANSFERS override") + } +} + +// unset removes a variable for the duration of the test, restoring it after. +func unset(t *testing.T, key string) { + t.Helper() + if old, ok := os.LookupEnv(key); ok { + t.Cleanup(func() { _ = os.Setenv(key, old) }) + } + _ = os.Unsetenv(key) +} diff --git a/internal/tdlkv/bolt.go b/internal/tdlkv/bolt.go new file mode 100644 index 0000000..9d1704c --- /dev/null +++ b/internal/tdlkv/bolt.go @@ -0,0 +1,140 @@ +// Package tdlkv opens the key-value store the tdl CLI writes, so this binary can +// reuse an existing `tdl login` session instead of introducing a second auth path. +// +// Layout, as implemented by tdl's bolt driver: the configured storage path is a +// directory, each namespace is a file inside it named after the namespace, and +// inside that file every value lives in a single bucket, also named after the +// namespace. So the default session is `/default`, bucket `default`. +// (The `data.kv` file some installs also carry belongs to tdl's older single-file +// driver and is not read here.) +// +// The store is opened read-write, not read-only, and that is not incidental: +// gotd rewrites the session blob whenever it establishes or re-establishes a +// connection, so an ordinary run does write here. Two consequences follow. +// First, bolt holds an exclusive file lock for as long as the store is open, so +// this binary and the `tdl` CLI cannot run against the same namespace at once — +// for a `doctor` that is a moment, for a long archive run it is the whole run. +// Second, the file is shared with a separately built binary: bbolt's on-disk +// format is stable across the versions involved (tdl pins v1.3.10, this module +// v1.5.0) and the default freelist type matches, so the sharing is safe, but a +// future bbolt major would need checking rather than assuming. +package tdlkv + +import ( + "context" + "errors" + "fmt" + "os" + "path/filepath" + "time" + + "go.etcd.io/bbolt" + + "github.com/iyear/tdl/core/storage" +) + +// lockTimeout bounds how long we wait for bolt's exclusive file lock. tdl holds +// that lock for its whole run, so without a timeout a concurrent `tdl` process +// makes this binary hang with no explanation. Three seconds is long enough to +// ride out a lock being handed over and short enough to fail fast. +const lockTimeout = 3 * time.Second + +// DefaultDir returns tdl's default storage directory. +func DefaultDir() string { + home, err := os.UserHomeDir() + if err != nil { + return ".tdl/data" + } + return filepath.Join(home, ".tdl", "data") +} + +// Store is a storage.Storage backed by one namespace of tdl's bolt store. +type Store struct { + db *bbolt.DB + bucket []byte +} + +// Open opens the namespace `ns` under the bolt directory `dir`. +// +// The namespace file must already exist: this binary never creates a session, +// it only reads the one `tdl login` produced. Creating it here would silently +// hand back an empty store and surface later as a confusing auth failure. +func Open(dir, ns string) (*Store, error) { + if ns == "" { + return nil, fmt.Errorf("namespace is required") + } + + path := filepath.Join(dir, ns) + if _, err := os.Stat(path); err != nil { + if os.IsNotExist(err) { + return nil, fmt.Errorf("no tdl session at %s — run `tdl login%s` first", + path, nsFlag(ns)) + } + return nil, fmt.Errorf("stat %s: %w", path, err) + } + + db, err := bbolt.Open(path, 0o600, &bbolt.Options{Timeout: lockTimeout}) + if err != nil { + if errors.Is(err, bbolt.ErrTimeout) { + return nil, fmt.Errorf("tdl session %s is locked by another process "+ + "(a running `tdl` or `tgexport`); stop it and retry", path) + } + return nil, fmt.Errorf("open %s: %w", path, err) + } + + s := &Store{db: db, bucket: []byte(ns)} + + // Fail here rather than on the first Get: a namespace file without its + // bucket is a store tdl never finished writing. + if err := db.View(func(tx *bbolt.Tx) error { + if tx.Bucket(s.bucket) == nil { + return fmt.Errorf("no bucket %q in %s — session looks incomplete, re-run `tdl login%s`", + ns, path, nsFlag(ns)) + } + return nil + }); err != nil { + _ = db.Close() + return nil, err + } + + return s, nil +} + +func nsFlag(ns string) string { + if ns == "default" { + return "" + } + return " -n " + ns +} + +func (s *Store) Get(_ context.Context, key string) ([]byte, error) { + var val []byte + if err := s.db.View(func(tx *bbolt.Tx) error { + // bbolt only guarantees a value is valid for the life of its + // transaction, so copy before returning it. + if v := tx.Bucket(s.bucket).Get([]byte(key)); v != nil { + val = append([]byte(nil), v...) + } + return nil + }); err != nil { + return nil, err + } + if val == nil { + return nil, storage.ErrNotFound + } + return val, nil +} + +func (s *Store) Set(_ context.Context, key string, value []byte) error { + return s.db.Update(func(tx *bbolt.Tx) error { + return tx.Bucket(s.bucket).Put([]byte(key), value) + }) +} + +func (s *Store) Delete(_ context.Context, key string) error { + return s.db.Update(func(tx *bbolt.Tx) error { + return tx.Bucket(s.bucket).Delete([]byte(key)) + }) +} + +func (s *Store) Close() error { return s.db.Close() } diff --git a/internal/tdlkv/bolt_test.go b/internal/tdlkv/bolt_test.go new file mode 100644 index 0000000..8476f49 --- /dev/null +++ b/internal/tdlkv/bolt_test.go @@ -0,0 +1,179 @@ +package tdlkv + +import ( + "context" + "errors" + "os" + "path/filepath" + "strings" + "testing" + + "go.etcd.io/bbolt" + + "github.com/iyear/tdl/core/storage" +) + +// newFixture writes a bolt file laid out the way tdl's driver lays one out: +// file named after the namespace, single bucket of the same name. +func newFixture(t *testing.T, dir, ns string, pairs map[string]string) { + t.Helper() + db, err := bbolt.Open(filepath.Join(dir, ns), 0o600, nil) + if err != nil { + t.Fatalf("create fixture: %v", err) + } + defer func() { _ = db.Close() }() + + if err := db.Update(func(tx *bbolt.Tx) error { + b, err := tx.CreateBucketIfNotExists([]byte(ns)) + if err != nil { + return err + } + for k, v := range pairs { + if err := b.Put([]byte(k), []byte(v)); err != nil { + return err + } + } + return nil + }); err != nil { + t.Fatalf("seed fixture: %v", err) + } +} + +func TestOpenReadsTdlLayout(t *testing.T) { + dir := t.TempDir() + newFixture(t, dir, "default", map[string]string{"app": "desktop"}) + + s, err := Open(dir, "default") + if err != nil { + t.Fatalf("Open: %v", err) + } + defer func() { _ = s.Close() }() + + got, err := s.Get(context.Background(), "app") + if err != nil { + t.Fatalf("Get: %v", err) + } + if string(got) != "desktop" { + t.Errorf("Get(app) = %q, want %q", got, "desktop") + } +} + +// A missing key must be distinguishable from an error, because core/storage +// consumers branch on ErrNotFound to fall back to defaults. +func TestGetMissingKeyReturnsErrNotFound(t *testing.T) { + dir := t.TempDir() + newFixture(t, dir, "default", nil) + + s, err := Open(dir, "default") + if err != nil { + t.Fatalf("Open: %v", err) + } + defer func() { _ = s.Close() }() + + if _, err := s.Get(context.Background(), "absent"); !errors.Is(err, storage.ErrNotFound) { + t.Errorf("Get(absent) error = %v, want storage.ErrNotFound", err) + } +} + +func TestSetAndDeleteRoundTrip(t *testing.T) { + dir := t.TempDir() + newFixture(t, dir, "default", nil) + + s, err := Open(dir, "default") + if err != nil { + t.Fatalf("Open: %v", err) + } + defer func() { _ = s.Close() }() + + ctx := context.Background() + if err := s.Set(ctx, "session", []byte("blob")); err != nil { + t.Fatalf("Set: %v", err) + } + got, err := s.Get(ctx, "session") + if err != nil || string(got) != "blob" { + t.Fatalf("Get after Set = %q, %v", got, err) + } + if err := s.Delete(ctx, "session"); err != nil { + t.Fatalf("Delete: %v", err) + } + if _, err := s.Get(ctx, "session"); !errors.Is(err, storage.ErrNotFound) { + t.Errorf("Get after Delete error = %v, want storage.ErrNotFound", err) + } +} + +// The three failure modes must stay distinguishable in the message, because +// each one tells the operator to do something different. +func TestOpenFailureMessages(t *testing.T) { + t.Run("missing namespace file", func(t *testing.T) { + _, err := Open(t.TempDir(), "default") + if err == nil { + t.Fatal("expected an error for a missing session file") + } + if !strings.Contains(err.Error(), "tdl login") { + t.Errorf("error should point at `tdl login`, got: %v", err) + } + }) + + t.Run("non-default namespace names the flag", func(t *testing.T) { + _, err := Open(t.TempDir(), "second") + if err == nil { + t.Fatal("expected an error for a missing session file") + } + if !strings.Contains(err.Error(), "-n second") { + t.Errorf("error should name the namespace flag, got: %v", err) + } + }) + + t.Run("file present but bucket missing", func(t *testing.T) { + dir := t.TempDir() + db, err := bbolt.Open(filepath.Join(dir, "default"), 0o600, nil) + if err != nil { + t.Fatalf("create empty db: %v", err) + } + _ = db.Close() + + if _, err := Open(dir, "default"); err == nil { + t.Fatal("expected an error for a bucketless store") + } else if !strings.Contains(err.Error(), "incomplete") { + t.Errorf("error should call the session incomplete, got: %v", err) + } + }) + + t.Run("empty namespace rejected", func(t *testing.T) { + if _, err := Open(t.TempDir(), ""); err == nil { + t.Fatal("expected an error for an empty namespace") + } + }) +} + +// Open must not create a session that does not exist: silently handing back an +// empty store would surface much later as a confusing auth failure. +func TestOpenDoesNotCreateMissingStore(t *testing.T) { + dir := t.TempDir() + if _, err := Open(dir, "default"); err == nil { + t.Fatal("expected an error") + } + if _, err := os.Stat(filepath.Join(dir, "default")); !os.IsNotExist(err) { + t.Errorf("Open created %s; it must never create a session file", filepath.Join(dir, "default")) + } +} + +// A second opener must be told the store is locked rather than hanging. +func TestOpenReportsLockContention(t *testing.T) { + dir := t.TempDir() + newFixture(t, dir, "default", nil) + + first, err := Open(dir, "default") + if err != nil { + t.Fatalf("first Open: %v", err) + } + defer func() { _ = first.Close() }() + + _, err = Open(dir, "default") + if err == nil { + t.Fatal("expected the second Open to fail while the first holds the lock") + } + if !strings.Contains(err.Error(), "locked by another process") { + t.Errorf("error should name lock contention, got: %v", err) + } +} diff --git a/internal/tgsource/client.go b/internal/tgsource/client.go new file mode 100644 index 0000000..72de17d --- /dev/null +++ b/internal/tgsource/client.go @@ -0,0 +1,152 @@ +// Package tgsource owns the Telegram side: building an authenticated client from +// a tdl session and pooling connections across data centres. +package tgsource + +import ( + "context" + "fmt" + "time" + + "github.com/gotd/td/telegram" + + "github.com/iyear/tdl/core/dcpool" + "github.com/iyear/tdl/core/storage" + "github.com/iyear/tdl/core/storage/keygen" + "github.com/iyear/tdl/core/tclient" +) + +// defaultReconnectTimeout matches tdl's own default. +const defaultReconnectTimeout = 5 * time.Minute + +// app describes the Telegram application a session was authorised against. +// +// A session is bound to the app that created it, so the AppID/AppHash used here +// must match the ones `tdl login` used or Telegram rejects the auth key. tdl +// records the choice in the store under the "app" key and defaults to its own +// application when the key is absent; mirror that exactly rather than hardcoding +// one, or a session created with `tdl login -d` fails to open. +type app struct { + id int + hash string +} + +var apps = map[string]app{ + // Application registered by tdl's author; tdl's default. + "builtin": {id: 15055931, hash: "021d433426cbb920eeb95164498fe3d3"}, + // Telegram Desktop's application, used by `tdl login -d`. + "desktop": {id: 2040, hash: "b18441a1ff607e10a989891a5462e627"}, +} + +// Options configures a session built from an existing tdl login. +type Options struct { + KV storage.Storage + Proxy string + NTP string + ReconnectTimeout time.Duration + PoolSize int64 +} + +// Session is an authenticated Telegram client together with the settings needed +// to build a DC pool over it. +// +// A gotd client cannot be reused: telegram.Client.Run refuses a second call once +// the first has returned. Anything that needs to retry a whole run must build a +// new Session rather than calling Run twice. +type Session struct { + client *telegram.Client + timeout time.Duration + poolSize int64 +} + +// New builds a Telegram session from the credentials in kv. +// +// Nothing connects yet: gotd dials inside Run, so callers must do their work in +// the callback Run provides. +func New(ctx context.Context, o Options) (*Session, error) { + a, err := resolveApp(ctx, o.KV) + if err != nil { + return nil, err + } + + if o.ReconnectTimeout == 0 { + o.ReconnectTimeout = defaultReconnectTimeout + } + if o.PoolSize == 0 { + o.PoolSize = 8 // tdl's default DC pool size + } + + // Middlewares are deliberately left empty here. core/tclient.New already + // prepends NewDefaultMiddlewares (recovery, retry, floodwait) to whatever it + // is given, so passing them again nests retry inside retry — three levels + // deep once core's own copy is counted, turning a hard RPC failure into + // minutes of silent backoff. The pool gets them explicitly in Run instead; + // see the comment there. + client, err := tclient.New(ctx, tclient.Options{ + AppID: a.id, + AppHash: a.hash, + Session: storage.NewSession(o.KV, false), + Proxy: o.Proxy, + NTP: o.NTP, + ReconnectTimeout: o.ReconnectTimeout, + }) + if err != nil { + return nil, err + } + + return &Session{client: client, timeout: o.ReconnectTimeout, poolSize: o.PoolSize}, nil +} + +// Client exposes the underlying client for calls that do not need the pool. +// Only valid inside a Run callback. +func (s *Session) Client() *telegram.Client { return s.client } + +func resolveApp(ctx context.Context, kv storage.Storage) (app, error) { + mode := "builtin" + if v, err := kv.Get(ctx, keygen.New("app")); err == nil { + mode = string(v) + } + a, ok := apps[mode] + if !ok { + return app{}, fmt.Errorf("session records unknown app %q; re-run `tdl login`", mode) + } + return a, nil +} + +// Run connects, verifies the session is authorised, and invokes fn with a DC +// pool. The pool is closed before Run returns. +func (s *Session) Run(ctx context.Context, fn func(context.Context, dcpool.Pool) error) error { + err := s.client.Run(ctx, func(ctx context.Context) error { + status, err := s.client.Auth().Status(ctx) + if err != nil { + return fmt.Errorf("auth status: %w", err) + } + if !status.Authorized { + return fmt.Errorf("tdl session is not authorized; run `tdl login`") + } + + // gotd applies a client's middlewares only to Client.Invoke. Calls made + // through a pooled DC connection bypass them entirely, so the pool needs + // its own copy or downloads run with no flood-wait handling and no retry + // — which on a multi-hour archive dies at the first FLOOD_WAIT. tdl does + // the same thing for the same reason (app/dl/dl.go). + pool := dcpool.NewPool(s.client, s.poolSize, + tclient.NewDefaultMiddlewares(ctx, s.timeout)...) + defer func() { _ = pool.Close() }() + + return fn(ctx, pool) + }) + + // gotd swallows cancellation: telegram.Client.Run ends with + // if err := g.Wait(); !errors.Is(err, context.Canceled) { return err } + // return nil + // so an interrupted run — and any callback error wrapping context.Canceled — + // comes back as success. Reporting that as exit 0 would tell a driver the + // archive is complete when it was abandoned half-way, which is exactly the + // failure the shell pipeline's `trap ... exit 130` existed to prevent. + if err == nil { + if cerr := ctx.Err(); cerr != nil { + return cerr + } + } + return err +} From b0c163ed87e83813f07d5747193e87b54168887d Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Sun, 6 Sep 2026 17:37:18 +0700 Subject: [PATCH 02/16] feat: resolve chats, walk history, and derive one canonical filename Second slice: the read path, and the fix for the bug that motivated the rewrite. The shell pipeline derived a filename twice. `tdl chat export` wrote the raw Telegram name into a JSON, while `tdl dl` rendered it through a template whose default pipes it through filenamify, which rewrites reserved characters and collapses runs of '!'. A message whose name contained '!!' was therefore looked up under one name and stored under another; the verifier never found it and re-fetched it on every pass. Here a single function produces the name, and the string it returns is used both to test for presence and to write the file, so the two cannot disagree. Names are stored exactly as Telegram reports them rather than reproducing filenamify. That is a deliberate break from what the old pipeline wrote: a file it stored under a rewritten name is not recognised and will be fetched again. For the one chat archived so far that is a single file out of 18155, already removed. Because names are verbatim they are not path-safe, so the code that turns one into a path must enforce containment. Walk yields a sequence rather than taking a callback, since the downloader consumes a pull iterator and range-over-func converts either way without anyone owning a goroutine. It pages newest first, where the old pipeline went oldest first, which changes what an interrupted run leaves behind. Message links are refused rather than guessed at, across every host Telegram uses and the tg:// forms that carry the message id in a query parameter. A private channel link and a public message link have the same shape, so the two are told apart by parsing rather than by pattern. Verified against the live chat: 18155 media messages, matching the shell verifier, and every one of the 15548 objects already on the remote is found under a derived name. --- cmd/tgexport/list.go | 85 ++++++++++++++++++ cmd/tgexport/main.go | 3 + internal/naming/naming.go | 62 +++++++++++++ internal/naming/naming_test.go | 89 +++++++++++++++++++ internal/naming/sole_source_test.go | 75 ++++++++++++++++ internal/tgsource/chat.go | 133 ++++++++++++++++++++++++++++ internal/tgsource/chat_test.go | 103 +++++++++++++++++++++ internal/tgsource/iterate.go | 86 ++++++++++++++++++ 8 files changed, 636 insertions(+) create mode 100644 cmd/tgexport/list.go create mode 100644 internal/naming/naming.go create mode 100644 internal/naming/naming_test.go create mode 100644 internal/naming/sole_source_test.go create mode 100644 internal/tgsource/chat.go create mode 100644 internal/tgsource/chat_test.go create mode 100644 internal/tgsource/iterate.go diff --git a/cmd/tgexport/list.go b/cmd/tgexport/list.go new file mode 100644 index 0000000..7afb020 --- /dev/null +++ b/cmd/tgexport/list.go @@ -0,0 +1,85 @@ +package main + +import ( + "bufio" + "context" + "errors" + "flag" + "fmt" + "os" + + "github.com/iyear/tdl/core/dcpool" + tdlstorage "github.com/iyear/tdl/core/storage" + + "github.com/tiennm99dev/telegram-exporter/internal/tdlkv" + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" +) + +// listCmd prints every media message in a chat as `idsizename`. +// +// It is the smallest thing that exercises the whole read path — resolve a chat, +// walk its history, derive a name — so a naming or paging problem shows up here +// rather than halfway through an archive run. The output is tab-separated on +// purpose: filenames contain spaces, commas and quotes, but not tabs. +func listCmd(ctx context.Context, args []string) error { + fs := flag.NewFlagSet("list", flag.ContinueOnError) + var ( + chat = fs.String("c", "", "chat id, username, or t.me link (required)") + ns = fs.String("n", "default", "tdl session namespace") + dataDir = fs.String("storage", tdlkv.DefaultDir(), "tdl bolt storage directory") + ) + if err := fs.Parse(args); err != nil { + if errors.Is(err, flag.ErrHelp) { + return err + } + return fmt.Errorf("%w: %v", errUsage, err) + } + if *chat == "" { + return fmt.Errorf("%w: -c CHAT is required", errUsage) + } + + kv, err := tdlkv.Open(*dataDir, *ns) + if err != nil { + return err + } + defer func() { _ = kv.Close() }() + + sess, err := tgsource.New(ctx, tgsource.Options{KV: kv}) + if err != nil { + return err + } + + return sess.Run(ctx, func(ctx context.Context, pool dcpool.Pool) error { + api := pool.Default(ctx) + + peer, err := tgsource.ResolveChat(ctx, tgsource.Manager(api, tdlstorage.NewPeers(kv)), *chat) + if err != nil { + return err + } + + // Buffered: one write syscall per item would dominate the runtime on a + // chat with tens of thousands of messages. Flushed explicitly below so a + // write failure — a closed pipe, a full disk — is reported rather than + // swallowed by a deferred call nobody checks. + out := bufio.NewWriter(os.Stdout) + + count := 0 + var total int64 + for it, err := range tgsource.Walk(ctx, api, peer) { + if err != nil { + return err + } + count++ + total += it.Size() + if _, err := fmt.Fprintf(out, "%d\t%d\t%s\n", it.MessageID, it.Size(), it.Name); err != nil { + return err + } + } + + if err := out.Flush(); err != nil { + return err + } + fmt.Fprintf(os.Stderr, "\n%d media messages, %.1f GiB\n", count, float64(total)/(1<<30)) + return nil + }) +} diff --git a/cmd/tgexport/main.go b/cmd/tgexport/main.go index cc9738c..eab509c 100644 --- a/cmd/tgexport/main.go +++ b/cmd/tgexport/main.go @@ -61,6 +61,8 @@ func run() int { switch os.Args[1] { case "doctor": err = doctorCmd(ctx, os.Args[2:]) + case "list": + err = listCmd(ctx, os.Args[2:]) default: fmt.Fprintf(os.Stderr, "unknown command %q\n\n", os.Args[1]) usage() @@ -147,6 +149,7 @@ func usage() { Commands: doctor Check the Telegram session, the destination remote, and free space + list Print every media message in a chat as idsizename Run 'tgexport -h' for command options. `) diff --git a/internal/naming/naming.go b/internal/naming/naming.go new file mode 100644 index 0000000..629ffe5 --- /dev/null +++ b/internal/naming/naming.go @@ -0,0 +1,62 @@ +// Package naming builds the filename a media message is stored under. +// +// This package exists to have exactly one answer to "what is this file called". +// The shell pipeline it replaces had two, and they disagreed. `tdl chat export` +// wrote the raw Telegram filename into a JSON, while `tdl dl` ran that same name +// through its download template — whose default is +// +// {{ .DialogID }}_{{ .MessageID }}_{{ filenamify .FileName }} +// +// (tdl@v0.20.4/cmd/dl.go:50). `filenamify` rewrites characters a filesystem +// rejects and, incidentally, collapses any run of two or more '!' into one. So a +// message whose filename contained "!!" was checked for under one name and +// stored under another; the verifier never found it and re-fetched it on every +// pass, forever. That is not a hypothetical — it cost 966 MB per pass on one +// message in this repo's own archive. +// +// The rule here is therefore not "be careful to keep the two in sync". There is +// one function, called once per message, and the string it returns is used both +// to ask whether the file is already archived and to write it. A divergence +// between those two questions is not made unlikely; it is made unrepresentable. +package naming + +import ( + "strconv" + "strings" + + "github.com/iyear/tdl/core/tmedia" +) + +// Separator between the three fields of a stored filename. +const sep = "_" + +// For returns the archive filename for one media message: +// +// {DialogID}_{MessageID}_{FileName} +// +// The field layout matches tdl's default template, but FileName does not: tdl +// passes it through `filenamify` and this does not. That is a deliberate, +// recorded choice (see the plan's Phase 2 notes) — names stay as Telegram +// reports them rather than being rewritten — and it means a file the old shell +// pipeline stored under a filenamify-altered name will not be recognised here +// and will be fetched again. For this repo's archive that affects exactly one +// message out of 12,000, and it was already removed. +// +// FileName comes from tmedia, the same extractor tdl uses: a document's +// DocumentAttributeFilename, or a generated stable name when it has none +// (`.jpg` for a photo, `` for a document). +// +// The name is taken verbatim and is therefore NOT safe to join onto a path. +// DocumentAttributeFilename is set by whoever uploaded the file, so it can +// contain '/' or '..' and escape a staging directory. Callers that turn a name +// into a path must check containment themselves; doing it here would silently +// rewrite names and reintroduce exactly the two-derivations problem above. +func For(dialogID int64, messageID int, m *tmedia.Media) string { + var b strings.Builder + b.WriteString(strconv.FormatInt(dialogID, 10)) + b.WriteString(sep) + b.WriteString(strconv.Itoa(messageID)) + b.WriteString(sep) + b.WriteString(m.Name) + return b.String() +} diff --git a/internal/naming/naming_test.go b/internal/naming/naming_test.go new file mode 100644 index 0000000..a96f8e4 --- /dev/null +++ b/internal/naming/naming_test.go @@ -0,0 +1,89 @@ +package naming + +import ( + "testing" + + "github.com/iyear/tdl/core/tmedia" +) + +// The format is a compatibility contract, not a style choice: ~15k files are +// already stored under it. Changing it makes every one of them look absent and +// re-downloads the entire archive. +func TestForMatchesStoredLayout(t *testing.T) { + tests := []struct { + name string + dialogID int64 + msgID int + file string + want string + }{ + { + name: "document with a filename attribute", + dialogID: 1234567890, + msgID: 14726, + file: "298.mp4", + want: "1234567890_14726_298.mp4", + }, + { + // The message that started this rewrite. Its name carries a doubled + // '!' which tdl's default template collapsed to one, via filenamify, + // while the export JSON kept both — the two derivations that never + // agreed. This package keeps the name as Telegram reports it, so the + // doubled '!' must survive. + name: "punctuation is preserved verbatim", + dialogID: 1234567890, + msgID: 4242, + file: "Pipe her!! And by her, we mean pipeperr! 1080p.mp4", + want: "1234567890_4242_Pipe her!! And by her, we mean pipeperr! 1080p.mp4", + }, + { + name: "photo gets tmedia's generated name", + dialogID: 1234567890, + msgID: 42, + file: "5901234567890123456.jpg", + want: "1234567890_42_5901234567890123456.jpg", + }, + { + name: "spaces and separators inside the filename are untouched", + dialogID: 1234567890, + msgID: 7, + file: "a_b c-d.e.mp4", + want: "1234567890_7_a_b c-d.e.mp4", + }, + { + name: "non-ascii is untouched", + dialogID: 1234567890, + msgID: 8, + file: "ünïcödé näme 🍓.mp4", + want: "1234567890_8_ünïcödé näme 🍓.mp4", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got := For(tt.dialogID, tt.msgID, &tmedia.Media{Name: tt.file}) + if got != tt.want { + t.Errorf("For() = %q, want %q", got, tt.want) + } + }) + } +} + +// Guards the reason the field layout is what it is. If this ever starts failing, +// tdl changed its default template and the compatibility note on For is stale. +func TestForKeepsCharactersTdlWouldRewrite(t *testing.T) { + // filenamify (tdl's default template applies it) collapses runs of '!' and + // replaces reserved characters. None of that may happen here. + for _, file := range []string{ + "double!!bang.mp4", + "a?b:c.mp4", + ".leading-dot.mp4", + "trailing!.mp4", + } { + got := For(1, 2, &tmedia.Media{Name: file}) + want := "1_2_" + file + if got != want { + t.Errorf("For(%q) = %q, want %q — a sanitiser crept in", file, got, want) + } + } +} diff --git a/internal/naming/sole_source_test.go b/internal/naming/sole_source_test.go new file mode 100644 index 0000000..fe7f743 --- /dev/null +++ b/internal/naming/sole_source_test.go @@ -0,0 +1,75 @@ +package naming + +import ( + "os" + "path/filepath" + "regexp" + "strconv" + "strings" + "testing" +) + +// buildsAName matches the ways a stored filename would plausibly be assembled by +// hand: the format verbs ("%d_%d_%s" and relatives) and string concatenation +// around a bare "_" separator. +// +// This is a lint for known shapes, not a proof. A determined reimplementation — +// a strings.Builder copy of For, say — still slips through. It catches the +// realistic accident, which is someone reaching for Sprintf in a new file. +var buildsAName = regexp.MustCompile(`%[ds]_%[ds]|_%[ds]_|\+ *"_" *\+`) + +// The whole point of this package is that it is the only answer to "what is this +// file called". Nothing in the type system prevents a second place from +// formatting the same string, so the common ways of doing so are linted here. +// +// If this test fails, the fix is to call For (or MessageID) rather than to widen +// the pattern. The shell pipeline's bug was two independent name derivations that +// nothing forced to agree; a second one here would reintroduce it exactly. +func TestNamingIsTheSoleSourceOfFilenames(t *testing.T) { + root, err := filepath.Abs("../..") + if err != nil { + t.Fatalf("locate repo root: %v", err) + } + + var offenders []string + err = filepath.WalkDir(root, func(path string, d os.DirEntry, err error) error { + if err != nil { + return err + } + if d.IsDir() { + // Skip VCS and anything vendored; only our own source counts. + switch d.Name() { + case ".git", "vendor", "plans", "staging": + return filepath.SkipDir + } + return nil + } + if !strings.HasSuffix(path, ".go") || strings.HasSuffix(path, "_test.go") { + return nil + } + // This package is the one place allowed to build the name. + if filepath.Dir(path) == filepath.Join(root, "internal", "naming") { + return nil + } + + src, err := os.ReadFile(path) + if err != nil { + return err + } + for i, line := range strings.Split(string(src), "\n") { + if buildsAName.MatchString(line) { + rel, _ := filepath.Rel(root, path) + offenders = append(offenders, rel+":"+strconv.Itoa(i+1)+": "+strings.TrimSpace(line)) + } + } + return nil + }) + if err != nil { + t.Fatalf("walk source tree: %v", err) + } + + if len(offenders) > 0 { + t.Errorf("filenames must only be built by naming.For; found %d other place(s):\n %s", + len(offenders), strings.Join(offenders, "\n ")) + } +} diff --git a/internal/tgsource/chat.go b/internal/tgsource/chat.go new file mode 100644 index 0000000..7729c84 --- /dev/null +++ b/internal/tgsource/chat.go @@ -0,0 +1,133 @@ +package tgsource + +import ( + "context" + "fmt" + "regexp" + "strings" + + "github.com/gotd/td/telegram/peers" + + "github.com/iyear/tdl/core/util/tutil" +) + +// botAPIID matches a Bot API chat id: the same channel as an MTProto id, but +// with a -100 prefix that MTProto itself does not use. +var botAPIID = regexp.MustCompile(`^-100(\d+)$`) + +// telegramHosts are the hosts Telegram deep links use. gotd accepts all three +// (telegram/deeplink/deeplink.go hasTelegramPrefix), so checking only t.me would +// let a telegram.me or telegram.dog message link through — and gotd's parser +// keeps just the domain and silently drops the message number, which would walk +// an entire chat when the operator asked for one message. +var telegramHosts = []string{"t.me/", "telegram.me/", "telegram.dog/"} + +// isMessageLink reports whether s points at a single message rather than a chat. +// +// This is parsed rather than pattern-matched because the two HTTPS shapes overlap +// in a way a regex gets wrong: a private link is t.me/c// and a public +// one is t.me//, so "t.me/c/1234567890" — a perfectly good private +// *channel* link — looks exactly like a public message link with the username +// "c". The distinction is whether a trailing numeric component follows the chat, +// and where that component sits depends on the "c" marker. +func isMessageLink(s string) bool { + lower := strings.ToLower(s) + + // tg:// links name the message in a query parameter rather than the path. + if strings.HasPrefix(lower, "tg://") { + for _, key := range []string{"post=", "message_id="} { + if strings.Contains(lower, "?"+key) || strings.Contains(lower, "&"+key) { + return true + } + } + return false + } + + rest, ok := afterHost(lower) + if !ok { + return false + } + rest, _, _ = strings.Cut(rest, "?") + rest, _, _ = strings.Cut(rest, "#") + + parts := strings.Split(strings.Trim(rest, "/"), "/") + if len(parts) >= 1 && parts[0] == "c" { + // c/ is the channel; c// is one message in it. + return len(parts) >= 3 && isDigits(parts[2]) + } + // t.me/joinchat/ and t.me/s/ are chats, not messages, and their + // second component is not a bare number — except for a hypothetical all-digit + // invite hash, which is not worth mis-parsing every real link to guard. + if len(parts) >= 1 && (parts[0] == "joinchat" || parts[0] == "s") { + return false + } + return len(parts) >= 2 && isDigits(parts[1]) +} + +// afterHost returns the path following a Telegram host, if s names one. +func afterHost(lower string) (string, bool) { + for _, host := range telegramHosts { + if i := strings.Index(lower, host); i >= 0 { + return lower[i+len(host):], true + } + } + return "", false +} + +func isDigits(s string) bool { + if s == "" { + return false + } + for _, r := range s { + if r < '0' || r > '9' { + return false + } + } + return true +} + +// NormalizeChat converts a chat argument into the form the resolver expects. +// +// The accepted forms are the ones the shell pipeline accepted, because they are +// what an operator already has to hand: a numeric MTProto id as printed by +// `tdl chat ls`, a username with or without '@', or a t.me/tg:// link. Two need +// help. A Bot API id carries a -100 prefix that MTProto does not use, and a +// message link is not a chat — silently treating one as a chat would export the +// wrong thing, so it is refused with an explanation rather than guessed at. +func NormalizeChat(chat string) (string, error) { + chat = strings.TrimSpace(chat) + if chat == "" { + return "", fmt.Errorf("a chat is required") + } + + if isMessageLink(chat) { + return "", fmt.Errorf("%q is a message link, not a chat — "+ + "pass the chat's username or id instead", chat) + } + + if m := botAPIID.FindStringSubmatch(chat); m != nil { + return m[1], nil + } + + // The resolver takes a bare username; '@' is how humans write it. + return strings.TrimPrefix(chat, "@"), nil +} + +// ResolveChat turns a chat argument into a peer. +// +// Numeric arguments are looked up as channel, then user, then chat ids; +// everything else goes through the resolver, which handles usernames and +// t.me/tg:// links. That ordering is tdl's (core/util/tutil.GetInputPeer), kept +// so an id that works in `tdl` works here. +func ResolveChat(ctx context.Context, manager *peers.Manager, chat string) (peers.Peer, error) { + normalized, err := NormalizeChat(chat) + if err != nil { + return nil, err + } + + peer, err := tutil.GetInputPeer(ctx, manager, normalized) + if err != nil { + return nil, fmt.Errorf("cannot resolve chat %q: %w", chat, err) + } + return peer, nil +} diff --git a/internal/tgsource/chat_test.go b/internal/tgsource/chat_test.go new file mode 100644 index 0000000..d13deff --- /dev/null +++ b/internal/tgsource/chat_test.go @@ -0,0 +1,103 @@ +package tgsource + +import ( + "strings" + "testing" +) + +// Every form the shell pipeline accepted must still be accepted, and the one +// form it refused must still be refused. An operator's existing command lines +// are the compatibility surface here. +func TestNormalizeChat(t *testing.T) { + tests := []struct { + name string + in string + want string + }{ + {"numeric mtproto id", "1234567890", "1234567890"}, + {"username with at", "@mychannel", "mychannel"}, + {"bare username", "mychannel", "mychannel"}, + {"public t.me link", "https://t.me/mychannel", "https://t.me/mychannel"}, + {"tg protocol link", "tg://resolve?domain=mychannel", "tg://resolve?domain=mychannel"}, + {"bot api id loses the -100 prefix", "-1001234567890", "1234567890"}, + {"surrounding whitespace is trimmed", " mychannel\n", "mychannel"}, + {"private channel link without a message", "https://t.me/c/1234567890", "https://t.me/c/1234567890"}, + // Invite and preview links are chats; their second component is not a + // bare message number and must not be read as one. + {"invite link", "https://t.me/+AbCd_1234", "https://t.me/+AbCd_1234"}, + {"joinchat link", "https://t.me/joinchat/AbCd1234", "https://t.me/joinchat/AbCd1234"}, + {"preview link", "https://t.me/s/mychannel", "https://t.me/s/mychannel"}, + {"other telegram host, no message", "https://telegram.dog/mychannel", "https://telegram.dog/mychannel"}, + {"tg link without a post parameter", "tg://resolve?domain=mychannel", "tg://resolve?domain=mychannel"}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got, err := NormalizeChat(tt.in) + if err != nil { + t.Fatalf("NormalizeChat(%q) returned an error: %v", tt.in, err) + } + if got != tt.want { + t.Errorf("NormalizeChat(%q) = %q, want %q", tt.in, got, tt.want) + } + }) + } +} + +// A message link names one message, not a chat. Guessing the chat from it would +// quietly export something the operator did not ask for, so it is refused. +func TestNormalizeChatRejectsMessageLinks(t *testing.T) { + for _, in := range []string{ + "https://t.me/c/1234567890/4242", + "t.me/c/1234567890/4242", + "https://t.me/mychannel/4242", + // gotd accepts all three Telegram hosts and its parser keeps only the + // domain, silently dropping the message number — so missing one of these + // would walk an entire chat when one message was asked for. + "https://telegram.me/mychannel/4242", + "https://telegram.dog/mychannel/4242", + "https://T.ME/mychannel/4242", + // tg:// names the message in a query parameter, not the path. + "tg://privatepost?channel=1234567890&post=4242", + "tg://resolve?domain=mychannel&post=4242", + "tg://openmessage?user_id=1&message_id=4242", + } { + t.Run(in, func(t *testing.T) { + _, err := NormalizeChat(in) + if err == nil { + t.Fatalf("NormalizeChat(%q) succeeded; a message link is not a chat", in) + } + if !strings.Contains(err.Error(), "message link") { + t.Errorf("error should say it is a message link, got: %v", err) + } + }) + } +} + +func TestNormalizeChatRejectsEmpty(t *testing.T) { + for _, in := range []string{"", " ", "\t\n"} { + if _, err := NormalizeChat(in); err == nil { + t.Errorf("NormalizeChat(%q) succeeded, want an error", in) + } + } +} + +// -100 is stripped only when it prefixes a Bot API id, never from an ordinary +// number that happens to start with those digits. +func TestNormalizeChatOnlyStripsRealBotAPIPrefix(t *testing.T) { + tests := map[string]string{ + "-1001234567890": "1234567890", // Bot API id + "1001234567890": "1001234567890", // no leading '-', not a Bot API id + "-100": "-100", // prefix with no id after it + "-2001234567890": "-2001234567890", // different prefix + } + for in, want := range tests { + got, err := NormalizeChat(in) + if err != nil { + t.Fatalf("NormalizeChat(%q): %v", in, err) + } + if got != want { + t.Errorf("NormalizeChat(%q) = %q, want %q", in, got, want) + } + } +} diff --git a/internal/tgsource/iterate.go b/internal/tgsource/iterate.go new file mode 100644 index 0000000..53d2d66 --- /dev/null +++ b/internal/tgsource/iterate.go @@ -0,0 +1,86 @@ +package tgsource + +import ( + "context" + "fmt" + "iter" + + "github.com/gotd/td/telegram/peers" + "github.com/gotd/td/telegram/query" + "github.com/gotd/td/tg" + + "github.com/iyear/tdl/core/tmedia" + + "github.com/tiennm99dev/telegram-exporter/internal/naming" +) + +// Item is one downloadable media message. +// +// Name is filled here, at the single point where the message is seen, and is the +// same string used to check the remote and to write the file. See the naming +// package for why that matters. Media carries the location, size and DC that the +// downloader needs, so nothing has to be looked up a second time. +type Item struct { + DialogID int64 + MessageID int + Name string + Media *tmedia.Media +} + +// Size reports the media size in bytes. +func (i Item) Size() int64 { return i.Media.Size } + +// Walk yields every media message in a chat, newest first. +// +// Order is Telegram's: GetHistory pages backwards from the most recent message. +// The shell pipeline fetched oldest-first, so an interrupted run leaves a +// different subset archived than the old one would have. +// +// A sequence rather than a callback because the downloader consumes a pull +// iterator (Next/Value/Err), and range-over-func converts either way for free: +// callers that want the callback shape just range over it, while iter.Pull2 +// gives the pull shape without anyone owning a goroutine. Messages are streamed, +// never collected — an 18k-message chat is tens of thousands of descriptors and +// the caller decides what to keep. +// +// Text-only and service messages carry no file and are skipped, the same rule +// the export JSON encoded as an empty "file" field. On error the sequence yields +// a zero Item with that error and stops; cancelling ctx stops it too, so an +// interrupted run does not keep paging. +func Walk(ctx context.Context, api *tg.Client, peer peers.Peer) iter.Seq2[Item, error] { + return func(yield func(Item, error) bool) { + dialogID := peer.ID() + + it := query.NewQuery(api).Messages().GetHistory(peer.InputPeer()).BatchSize(100).Iter() + for it.Next(ctx) { + msg, ok := it.Value().Msg.(*tg.Message) + if !ok { + continue // service messages have no media + } + + media, ok := tmedia.GetMedia(msg) + if !ok { + continue // text-only, or a media kind tmedia cannot download + } + + if !yield(Item{ + DialogID: dialogID, + MessageID: msg.ID, + Name: naming.For(dialogID, msg.ID, media), + Media: media, + }, nil) { + return + } + } + + if err := it.Err(); err != nil { + yield(Item{}, fmt.Errorf("walk chat history: %w", err)) + } + } +} + +// Manager builds a peers manager over the session's peer cache, so resolving the +// same chat twice does not cost a second round trip. +func Manager(api *tg.Client, storage peers.Storage) *peers.Manager { + return peers.Options{Storage: storage}.Build(api) +} From 1393abdae22cc9a3605a96b248c972b6fa4d2694 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Sun, 6 Sep 2026 18:50:53 +0700 Subject: [PATCH 03/16] feat: index a remote and verify a chat against it in-process MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Third slice: verify-export.sh, without the subprocess or the python. One rclone listing builds an in-memory index, and the report is computed from it. missing-ids.txt and gap.json are gone; so is every python3 heredoc. Presence is answered from a whole name and never from a message id. The id-keyed map exists only to tell "absent" apart from "absent, but a stale copy under an older name is sitting there", and it stays unexported so nothing can reach for it as an answer. That distinction is the bug this rewrite exists to remove, so it is enforced by structure rather than by comment. Names are checked for path containment before use. Storing them verbatim means a filename chosen by whoever uploaded the file can contain a separator or a parent reference, and tdl never had to care because its template rewrote those away. Over-long names are refused for the same reason: the filesystem would reject them at create time, and a file that can never be written would be reported absent on every pass forever. Objects are addressed by the path rclone knows them by, not by the basename used for matching. The two differ once a remote has directory structure, and deleting by basename would miss the object or remove a same-named one from the root. Basenames appearing at more than one path make the snapshot ambiguous, so they are reported rather than silently resolved. Indexing runs at full depth with filters cleared. Inheriting RCLONE_MAX_DEPTH or RCLONE_EXCLUDE would not fail, it would quietly report archived files as absent and fetch them all again. Deleting stale copies stays opt-in and confirmed; a non-interactive stdin declines rather than proceeding. Filenames are quoted wherever they are printed, so an embedded escape cannot redraw the list an operator approves. Verified against the live remote: identical to verify-export.sh on the same state — 18155 expected, 15548 present, 2607 absent, ids 9857-18013, exit 1. --- cmd/tgexport/main.go | 3 + cmd/tgexport/verify.go | 201 +++++++++++++++++++++++++++++++++ internal/naming/naming_test.go | 3 + internal/naming/safe.go | 92 +++++++++++++++ internal/naming/safe_test.go | 147 ++++++++++++++++++++++++ internal/pipeline/elem.go | 131 +++++++++++++++++++++ internal/remote/index.go | 134 ++++++++++++++++++++++ internal/remote/index_test.go | 174 ++++++++++++++++++++++++++++ internal/verify/verify.go | 176 +++++++++++++++++++++++++++++ internal/verify/verify_test.go | 181 +++++++++++++++++++++++++++++ 10 files changed, 1242 insertions(+) create mode 100644 cmd/tgexport/verify.go create mode 100644 internal/naming/safe.go create mode 100644 internal/naming/safe_test.go create mode 100644 internal/pipeline/elem.go create mode 100644 internal/remote/index.go create mode 100644 internal/remote/index_test.go create mode 100644 internal/verify/verify.go create mode 100644 internal/verify/verify_test.go diff --git a/cmd/tgexport/main.go b/cmd/tgexport/main.go index eab509c..0a515fe 100644 --- a/cmd/tgexport/main.go +++ b/cmd/tgexport/main.go @@ -63,6 +63,8 @@ func run() int { err = doctorCmd(ctx, os.Args[2:]) case "list": err = listCmd(ctx, os.Args[2:]) + case "verify": + err = verifyCmd(ctx, os.Args[2:]) default: fmt.Fprintf(os.Stderr, "unknown command %q\n\n", os.Args[1]) usage() @@ -150,6 +152,7 @@ func usage() { Commands: doctor Check the Telegram session, the destination remote, and free space list Print every media message in a chat as idsizename + verify Report whether a chat is fully archived on a remote Run 'tgexport -h' for command options. `) diff --git a/cmd/tgexport/verify.go b/cmd/tgexport/verify.go new file mode 100644 index 0000000..d79d1e7 --- /dev/null +++ b/cmd/tgexport/verify.go @@ -0,0 +1,201 @@ +package main + +import ( + "bufio" + "context" + "errors" + "flag" + "fmt" + "os" + "strings" + + "github.com/iyear/tdl/core/dcpool" + tdlstorage "github.com/iyear/tdl/core/storage" + rclonefs "github.com/rclone/rclone/fs" + "github.com/rclone/rclone/fs/operations" + + "github.com/tiennm99dev/telegram-exporter/internal/remote" + "github.com/tiennm99dev/telegram-exporter/internal/tdlkv" + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" + "github.com/tiennm99dev/telegram-exporter/internal/verify" +) + +// verifyCmd reports whether a chat is fully archived on a remote. +// +// Exit 0 means complete, 1 means it ran and found work outstanding. Those are +// distinct on purpose: a driver needs to tell "nothing left to do" from "still +// incomplete" without parsing output. +func verifyCmd(ctx context.Context, args []string) error { + fs := flag.NewFlagSet("verify", flag.ContinueOnError) + var ( + chat = fs.String("c", "", "chat id, username, or t.me link (required)") + remoteArg = fs.String("r", "", "rclone destination, e.g. pikpak:archive (required)") + ns = fs.String("n", "default", "tdl session namespace") + dataDir = fs.String("storage", tdlkv.DefaultDir(), "tdl bolt storage directory") + delStale = fs.Bool("delete-misnamed", false, "delete remote files stored under a superseded name") + assumeYes = fs.Bool("y", false, "do not prompt before deleting") + ) + if err := fs.Parse(args); err != nil { + if errors.Is(err, flag.ErrHelp) { + return err + } + return fmt.Errorf("%w: %v", errUsage, err) + } + if *chat == "" || *remoteArg == "" { + return fmt.Errorf("%w: -c CHAT and -r REMOTE:PATH are both required", errUsage) + } + + ctx, err := remote.Init(ctx, remote.DefaultTunables()) + if err != nil { + return err + } + dst, err := remote.Resolve(ctx, *remoteArg) + if err != nil { + return err + } + + kv, err := tdlkv.Open(*dataDir, *ns) + if err != nil { + return err + } + defer func() { _ = kv.Close() }() + + sess, err := tgsource.New(ctx, tgsource.Options{KV: kv}) + if err != nil { + return err + } + + var ( + report verify.Report + idx *remote.Index + ) + if err := sess.Run(ctx, func(ctx context.Context, pool dcpool.Pool) error { + api := pool.Default(ctx) + + peer, err := tgsource.ResolveChat(ctx, tgsource.Manager(api, tdlstorage.NewPeers(kv)), *chat) + if err != nil { + return err + } + + // The chat is walked first and the remote listed second, so the snapshot + // is never older than the wanted set. The reverse order could report a + // file absent that was uploaded while the walk was still running. + var items []tgsource.Item + for it, err := range tgsource.Walk(ctx, api, peer) { + if err != nil { + return err + } + items = append(items, it) + } + + idx, err = remote.BuildIndex(ctx, dst, peer.ID()) + if err != nil { + return err + } + fmt.Fprintf(os.Stderr, "indexed %d objects on %s\n", idx.Len(), dst.String()) + if dup := idx.Collisions(); len(dup) > 0 { + // An ambiguous snapshot makes every verdict about these names + // unreliable, so it is reported rather than silently resolved. + fmt.Fprintf(os.Stderr, "warning: %d basename(s) appear at more than one path; "+ + "verdicts for them may flip between runs:\n", len(dup)) + for _, d := range dup[:min(5, len(dup))] { + fmt.Fprintf(os.Stderr, " %q\n", d) + } + } + fmt.Fprintln(os.Stderr) + + report = verify.Check(items, idx) + return nil + }); err != nil { + return err + } + + out := bufio.NewWriter(os.Stdout) + report.Write(out) + if err := out.Flush(); err != nil { + return err + } + + if *delStale { + if err := deleteMisnamed(ctx, dst, idx, report, *assumeYes); err != nil { + return err + } + } + + if !report.Complete() { + return fmt.Errorf("%w: %d file(s) still to fetch", errIncomplete, len(report.Todo())) + } + return nil +} + +// deleteMisnamed removes stale copies left by an earlier naming scheme. +// +// Off by default and confirmed by default: this deletes data from the operator's +// remote, and a stale copy costs storage rather than correctness, so there is no +// hurry that justifies doing it unasked. +// +// Objects are addressed by the path the index recorded, not by the name matching +// used elsewhere. Those differ the moment a remote has directory structure, and +// deleting by basename would either miss the object or — worse, if the root +// happens to hold a same-named file — delete the wrong one. +func deleteMisnamed(ctx context.Context, dst rclonefs.Fs, idx *remote.Index, report verify.Report, assumeYes bool) error { + var targets []string + for _, m := range report.Misnamed { + targets = append(targets, m.Found...) + } + if len(targets) == 0 { + fmt.Fprintln(os.Stderr, "\nnothing to delete: no files stored under a superseded name") + return nil + } + + // Filenames are chosen by whoever uploaded the file and may contain control + // characters or bidi marks, so they are quoted rather than printed raw: an + // embedded newline or escape sequence could otherwise redraw this list and + // have the operator approve something other than what they read. + fmt.Fprintf(os.Stderr, "\nabout to delete %d file(s) from %s:\n", len(targets), dst.String()) + for _, t := range targets { + fmt.Fprintf(os.Stderr, " %q\n", t) + } + + if !assumeYes { + // Anything that is not an explicit yes leaves the files alone, and that + // includes the read failing. Closed or non-interactive stdin — cron, a + // pipeline — therefore declines rather than proceeding, which is the + // safe direction for a delete. + fmt.Fprint(os.Stderr, "delete these? [y/N] ") + var answer string + _, _ = fmt.Scanln(&answer) + switch strings.ToLower(strings.TrimSpace(answer)) { + case "y", "yes": + default: + fmt.Fprintln(os.Stderr, "left alone") + return nil + } + } + + // One failure must not strand the rest: the operator approved a set, so the + // whole set is attempted and the outcome reported as a count they can check + // against what they approved. + var errs []error + deleted := 0 + for _, name := range targets { + path, ok := idx.PathOf(name) + if !ok { + errs = append(errs, fmt.Errorf("%q is no longer in the index", name)) + continue + } + obj, err := dst.NewObject(ctx, path) + if err != nil { + errs = append(errs, fmt.Errorf("locate %q: %w", path, err)) + continue + } + if err := operations.DeleteFile(ctx, obj); err != nil { + errs = append(errs, fmt.Errorf("delete %q: %w", path, err)) + continue + } + deleted++ + } + + fmt.Fprintf(os.Stderr, "deleted %d of %d\n", deleted, len(targets)) + return errors.Join(errs...) +} diff --git a/internal/naming/naming_test.go b/internal/naming/naming_test.go index a96f8e4..a14d86c 100644 --- a/internal/naming/naming_test.go +++ b/internal/naming/naming_test.go @@ -87,3 +87,6 @@ func TestForKeepsCharactersTdlWouldRewrite(t *testing.T) { } } } + +// mediaNamed builds the only part of tmedia.Media these tests care about. +func mediaNamed(name string) *tmedia.Media { return &tmedia.Media{Name: name} } diff --git a/internal/naming/safe.go b/internal/naming/safe.go new file mode 100644 index 0000000..feebf6f --- /dev/null +++ b/internal/naming/safe.go @@ -0,0 +1,92 @@ +package naming + +import ( + "fmt" + "path/filepath" + "strconv" + "strings" +) + +// maxNameBytes is NAME_MAX on Linux: the longest single path component ext4 and +// friends accept. It is a byte count, not a rune count. +const maxNameBytes = 255 + +// Safe reports whether a stored name can be joined onto a directory path. +// +// Names are kept exactly as Telegram reports them, and the filename part comes +// from DocumentAttributeFilename — an unconstrained UTF-8 string chosen by +// whoever uploaded the file. So a name may contain '/' or be "..", and +// filepath.Join would happily resolve either outside the staging directory. tdl +// never had to think about this because its default template runs the name +// through filenamify, which rewrites separators and leading dots away; storing +// names verbatim moves that responsibility here. +// +// The rule is deliberately strict rather than corrective: a name that is not a +// single path element is rejected, not rewritten. Rewriting is what produced two +// disagreeing derivations in the first place, and a rejected file is a visible +// problem where a silently renamed one is not. +func Safe(name string) error { + switch { + case name == "": + return fmt.Errorf("empty filename") + case len(name) > maxNameBytes: + // The one rejection that is about a limit rather than an escape, and the + // one that matters most: os.Create returns ENAMETOOLONG past this, so an + // over-long name would download-fail forever while verify kept reporting + // it absent — a loop that never terminates. tdl could not hit this + // because filenamify truncates to 100 runes; storing names verbatim + // removes that cap, so the limit has to be checked instead. + return fmt.Errorf("filename is %d bytes, over the %d-byte limit: %q", + len(name), maxNameBytes, name) + case strings.ContainsRune(name, 0): + return fmt.Errorf("filename contains a NUL byte: %q", name) + case name == "." || name == "..": + return fmt.Errorf("filename is a directory reference: %q", name) + case filepath.IsAbs(name): + return fmt.Errorf("filename is an absolute path: %q", name) + case name != filepath.Base(name): + // Catches embedded separators, trailing slashes, and any ".." segment, + // since Base of all of those differs from the original. + return fmt.Errorf("filename is not a single path element: %q", name) + } + return nil +} + +// SplitStored recovers the message id from a stored name, reporting false when +// the name does not have the expected shape or belongs to another dialog. +// +// Diagnostics only — telling "absent" apart from "present under a different +// name" when reporting on a remote. It must never decide that the wanted file is +// present: a file whose id matches but whose name does not is a different file. +// remote.Index keeps its id-keyed map unexported and its only id-to-name route +// excludes the wanted name, so an id cannot yield "the file you asked for is +// here" — though a caller that deliberately looks up a different name will of +// course get an answer about that name. +func SplitStored(dialogID int64, name string) (messageID int, ok bool) { + prefix := strconv.FormatInt(dialogID, 10) + sep + rest, found := strings.CutPrefix(name, prefix) + if !found { + return 0, false + } + idStr, _, found := strings.Cut(rest, sep) + if !found { + return 0, false + } + // Reject anything Atoi would accept but For would never emit: a sign, or + // leading zeros. The id has to be the exact text For wrote. + if idStr == "" || idStr[0] == '0' { + return 0, false + } + for _, r := range idStr { + if r < '0' || r > '9' { + return 0, false + } + } + // Only an overflow can fail here: the loop above rejected non-digits and a + // leading zero, so anything that parses is already >= 1. + id, err := strconv.Atoi(idStr) + if err != nil { + return 0, false + } + return id, true +} diff --git a/internal/naming/safe_test.go b/internal/naming/safe_test.go new file mode 100644 index 0000000..531202d --- /dev/null +++ b/internal/naming/safe_test.go @@ -0,0 +1,147 @@ +package naming + +import ( + "path/filepath" + "strings" + "testing" +) + +// Names come from DocumentAttributeFilename, which whoever uploaded the file +// chose. Anything that is not a single path element must be refused before it +// reaches filepath.Join. +func TestSafeRejectsNamesThatEscapeADirectory(t *testing.T) { + bad := []struct { + name string + want string + }{ + {"", "empty"}, + {".", "directory reference"}, + {"..", "directory reference"}, + {"../escape.mp4", "single path element"}, + {"../../../.config/rclone/rclone.conf", "single path element"}, + {"sub/dir.mp4", "single path element"}, + {"trailing/", "single path element"}, + {"/etc/passwd", "absolute path"}, + {"/", "absolute path"}, + {"nul\x00byte.mp4", "NUL byte"}, + } + + for _, tt := range bad { + t.Run(tt.name, func(t *testing.T) { + err := Safe(tt.name) + if err == nil { + t.Fatalf("Safe(%q) = nil; this name escapes or breaks a path", tt.name) + } + if !strings.Contains(err.Error(), tt.want) { + t.Errorf("Safe(%q) error = %v, want it to mention %q", tt.name, err, tt.want) + } + }) + } +} + +// Ordinary media names, including awkward but legal ones, must pass — a +// containment check that rejects real files is just a different outage. +func TestSafeAcceptsRealNames(t *testing.T) { + for _, name := range []string{ + "1234567890_4242_Pipe her!! And by her, we mean pipeperr! 1080p.mp4", + "1234567890_14726_298.mp4", + "1234567890_8_ünïcödé näme 🍓.mp4", + "1234567890_42_a?b:c.mp4", // reserved on Windows, fine here + "1234567890_7_...dots.mp4", // leading dots inside the field, not the name + "1234567890_9_-dash.mp4", + "1234567890_10_ leading-space.mp4", + } { + if err := Safe(name); err != nil { + t.Errorf("Safe(%q) = %v, want nil", name, err) + } + } +} + +// The property that actually matters: a name Safe accepts cannot, once joined, +// resolve outside the directory it was joined to. +func TestSafeNamesStayInsideTheStagingDirectory(t *testing.T) { + const staging = "/var/tmp/staging" + for _, name := range []string{ + "1234567890_1_ok.mp4", + "1234567890_2_..dots.mp4", + "1234567890_3_a..b.mp4", + } { + if err := Safe(name); err != nil { + t.Fatalf("Safe(%q) = %v, want nil", name, err) + } + joined := filepath.Clean(filepath.Join(staging, name)) + if filepath.Dir(joined) != staging { + t.Errorf("Join(%q, %q) = %q, which leaves the staging directory", staging, name, joined) + } + } +} + +func TestSplitStoredRoundTripsWhatForProduces(t *testing.T) { + const dialog = int64(1234567890) + for _, msgID := range []int{1, 42, 4242, 4246} { + name := For(dialog, msgID, mediaNamed("a_b!!.mp4")) + got, ok := SplitStored(dialog, name) + if !ok { + t.Errorf("SplitStored(%q) reported no match", name) + continue + } + if got != msgID { + t.Errorf("SplitStored(%q) = %d, want %d", name, got, msgID) + } + } +} + +func TestSplitStoredRejectsForeignNames(t *testing.T) { + const dialog = int64(1234567890) + for _, name := range []string{ + "999_42_other-dialog.mp4", // different dialog + "1234567890_notanumber_.mp4", // id is not a number + "1234567890_0_zero.mp4", // ids start at 1 + "1234567890_042_pad.mp4", // For never emits leading zeros + "1234567890_+42_sign.mp4", // nor a sign + "1234567890_-5_neg.mp4", + "1234567890_42", // no filename field + "1234567890", // no id field + "", + } { + if id, ok := SplitStored(dialog, name); ok { + t.Errorf("SplitStored(%q) = %d, true; want no match", name, id) + } + } +} + +// A dialog id that is a prefix of another must not match it. +func TestSplitStoredDoesNotMatchPrefixOverlap(t *testing.T) { + if id, ok := SplitStored(123456789, "1234567890_42_file.mp4"); ok { + t.Errorf("SplitStored matched a longer dialog id, got %d", id) + } +} + +// An over-long name is the one input that can hang a drive-until-complete loop: +// os.Create rejects it, so the download fails forever while verify keeps +// reporting it absent. It must be refused up front, not discovered per pass. +func TestSafeRejectsNamesOverTheFilesystemLimit(t *testing.T) { + prefix := "1234567890_42_" + fill := 255 - len(prefix) + + atLimit := prefix + strings.Repeat("a", fill) + if err := Safe(atLimit); err != nil { + t.Errorf("Safe(%d bytes) = %v, want nil at exactly the limit", len(atLimit), err) + } + + overLimit := prefix + strings.Repeat("a", fill+1) + err := Safe(overLimit) + if err == nil { + t.Fatalf("Safe(%d bytes) = nil, want an error past the limit", len(overLimit)) + } + if !strings.Contains(err.Error(), "over the 255-byte limit") { + t.Errorf("error should name the limit, got: %v", err) + } + + // The limit is bytes, not runes: multi-byte names hit it sooner. + multibyte := prefix + strings.Repeat("é", 130) // 260 bytes of payload + if err := Safe(multibyte); err == nil { + t.Errorf("Safe(%d bytes, %d runes) = nil; the limit must count bytes", + len(multibyte), len([]rune(multibyte))) + } +} diff --git a/internal/pipeline/elem.go b/internal/pipeline/elem.go new file mode 100644 index 0000000..30e39fa --- /dev/null +++ b/internal/pipeline/elem.go @@ -0,0 +1,131 @@ +// Package pipeline drives the download half of an archive run. +package pipeline + +import ( + "context" + "fmt" + "io" + "iter" + "os" + "path/filepath" + + "github.com/gotd/td/tg" + + "github.com/iyear/tdl/core/downloader" + + "github.com/tiennm99dev/telegram-exporter/internal/naming" + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" +) + +// partSuffix marks a download that is still in flight. +// +// Deliberately not tdl's ".tmp": nothing here excludes by extension any more, +// because upload is triggered by a download returning rather than by a filter +// over a directory. A distinct suffix just keeps a staging directory shared with +// a legacy tdl run unambiguous during the cutover. +const partSuffix = ".part" + +// elem adapts one media item to the downloader's element interface. +type elem struct { + item tgsource.Item + file *os.File + takeout bool +} + +func (e *elem) File() downloader.File { return mediaFile{e.item} } +func (e *elem) To() io.WriterAt { return e.file } +func (e *elem) AsTakeout() bool { return e.takeout } + +// mediaFile exposes what the downloader needs to locate the bytes. All three +// values come straight from tmedia, so nothing is looked up a second time. +type mediaFile struct{ item tgsource.Item } + +func (f mediaFile) Location() tg.InputFileLocationClass { return f.item.Media.InputFileLoc } +func (f mediaFile) Size() int64 { return f.item.Media.Size } +func (f mediaFile) DC() int { return f.item.Media.DC } + +// partPath and finalPath are where an item is written and where it lands. +func partPath(staging string, it tgsource.Item) string { + return filepath.Join(staging, it.Name+partSuffix) +} +func finalPath(staging string, it tgsource.Item) string { + return filepath.Join(staging, it.Name) +} + +// elemIter turns the item sequence into the pull iterator the downloader wants, +// opening each destination file as it goes. +// +// The downloader consumes Next/Value/Err; Walk produces an iter.Seq2. iter.Pull2 +// bridges them without this code owning a goroutine or a channel, which is why +// Walk returns a sequence in the first place. +type elemIter struct { + next func() (tgsource.Item, error, bool) + stop func() + staging string + takeout bool + + current *elem + err error + + // opened records every file handle so a run can close them all. The + // downloader never closes what To() hands it, and a leak here is thousands + // of descriptors on a full archive run. + opened []*os.File +} + +func newElemIter(seq iter.Seq2[tgsource.Item, error], staging string, takeout bool) *elemIter { + next, stop := iter.Pull2(seq) + return &elemIter{next: next, stop: stop, staging: staging, takeout: takeout} +} + +func (i *elemIter) Next(ctx context.Context) bool { + if i.err != nil { + return false + } + if err := ctx.Err(); err != nil { + i.err = err + return false + } + + item, err, ok := i.next() + if !ok { + return false + } + if err != nil { + i.err = err + return false + } + + // A name that cannot be written is refused here rather than left to + // os.Create: the error names the message, and the run continues instead of + // failing on a path that could never have worked. + if err := naming.Safe(item.Name); err != nil { + i.err = fmt.Errorf("message %d: %w", item.MessageID, err) + return false + } + + f, err := os.OpenFile(partPath(i.staging, item), os.O_CREATE|os.O_RDWR, 0o600) + if err != nil { + i.err = fmt.Errorf("open destination for message %d: %w", item.MessageID, err) + return false + } + i.opened = append(i.opened, f) + + i.current = &elem{item: item, file: f, takeout: i.takeout} + return true +} + +func (i *elemIter) Value() downloader.Elem { return i.current } +func (i *elemIter) Err() error { return i.err } + +// Close releases the pull iterator and every file the walk opened. +func (i *elemIter) Close() error { + i.stop() + var firstErr error + for _, f := range i.opened { + if err := f.Close(); err != nil && firstErr == nil { + firstErr = err + } + } + return firstErr +} diff --git a/internal/remote/index.go b/internal/remote/index.go new file mode 100644 index 0000000..73bd950 --- /dev/null +++ b/internal/remote/index.go @@ -0,0 +1,134 @@ +package remote + +import ( + "context" + "fmt" + "path" + "slices" + + "github.com/rclone/rclone/fs" + "github.com/rclone/rclone/fs/filter" + "github.com/rclone/rclone/fs/operations" + + "github.com/tiennm99dev/telegram-exporter/internal/naming" +) + +// object is what the index remembers about one stored file. +// +// Path is kept alongside the basename because acting on an object — deleting a +// stale copy, say — needs the path rclone knows it by, while matching needs the +// basename. Conflating the two makes a delete address the wrong file, or no file +// at all, whenever a remote has any directory structure. +type object struct { + path string + size int64 +} + +// Index is a snapshot of what a remote holds, keyed by filename. +// +// It answers one question — "is this exact name present, and how big is it?" — +// and it answers it from the whole name, never from a message id. That is the +// point: a file whose id matches but whose name does not is a different file, +// and treating it as present is precisely the bug this rewrite exists to remove. +// +// The id-keyed map below exists only to tell "absent" apart from "absent, but +// something else is stored under this message's id" when reporting. It is +// unexported, and the only exported route from an id to a name is +// StoredUnderOtherNames, which by construction excludes the wanted name — so no +// caller can get "the file you asked for is present" out of an id. +type Index struct { + byName map[string]object + byID map[int][]string + collisions []string +} + +// BuildIndex lists a remote once and indexes it. +// +// Object paths are reduced to their basename, so a remote written with a +// subdirectory layout matches the same way a flat one does — the shell verifier +// did this too, and existing archives rely on it. +// +// Listing runs with depth and filters neutralised. rclone's ListFn otherwise +// inherits whatever RCLONE_MAX_DEPTH or RCLONE_EXCLUDE happen to be set to, and +// a narrowed listing here does not fail — it silently reports archived files as +// absent and re-downloads every one of them. The transfer tunables in Init are +// deliberately env-overridable; this is not. +func BuildIndex(ctx context.Context, f fs.Fs, dialogID int64) (*Index, error) { + ctx, ci := fs.AddConfig(ctx) + ci.MaxDepth = -1 + + unfiltered, err := filter.NewFilter(nil) + if err != nil { + return nil, fmt.Errorf("build an empty filter: %w", err) + } + ctx = filter.ReplaceConfig(ctx, unfiltered) + + idx := &Index{ + byName: make(map[string]object), + byID: make(map[int][]string), + } + + // ListFn is documented not to call fn concurrently, so the maps need no lock. + if err := operations.ListFn(ctx, f, func(o fs.Object) { + name := path.Base(o.Remote()) + + if _, seen := idx.byName[name]; seen { + // Two objects in different directories sharing a basename. Listing + // order is not guaranteed, so silently keeping one would make the + // verdict flip between runs — a zero-byte copy and a complete one + // would alternate. Keep the first and report the ambiguity instead. + if !slices.Contains(idx.collisions, name) { + idx.collisions = append(idx.collisions, name) + } + return + } + + idx.byName[name] = object{path: o.Remote(), size: o.Size()} + if id, ok := naming.SplitStored(dialogID, name); ok { + idx.byID[id] = append(idx.byID[id], name) + } + }); err != nil { + return nil, fmt.Errorf("list %s: %w", f.String(), err) + } + return idx, nil +} + +// Lookup reports the size stored under an exact name. +func (i *Index) Lookup(name string) (size int64, ok bool) { + o, ok := i.byName[name] + return o.size, ok +} + +// PathOf returns the remote path an indexed name was found at, which is what +// rclone needs to act on the object. It differs from the name whenever the +// remote has directory structure. +func (i *Index) PathOf(name string) (string, bool) { + o, ok := i.byName[name] + return o.path, ok +} + +// Len reports how many distinct names the remote held when the snapshot was +// taken. Objects dropped as basename collisions are not counted. +func (i *Index) Len() int { return len(i.byName) } + +// Collisions lists basenames that appeared at more than one path. A non-empty +// result means the snapshot is ambiguous and any verdict about those names is +// unreliable, so callers should surface it rather than ignore it. +func (i *Index) Collisions() []string { return i.collisions } + +// StoredUnderOtherNames lists names present for a message id that are not the +// wanted name. +// +// Diagnostics only. A non-empty result never means the file is archived — it +// means a stale copy from an earlier naming scheme is sitting there and will +// still be sitting there after the re-download, which is why the report has to +// surface it rather than quietly counting it. +func (i *Index) StoredUnderOtherNames(messageID int, wanted string) []string { + var others []string + for _, n := range i.byID[messageID] { + if n != wanted { + others = append(others, n) + } + } + return others +} diff --git a/internal/remote/index_test.go b/internal/remote/index_test.go new file mode 100644 index 0000000..5abd8c4 --- /dev/null +++ b/internal/remote/index_test.go @@ -0,0 +1,174 @@ +package remote + +import ( + "context" + "os" + "path/filepath" + "testing" + + _ "github.com/rclone/rclone/backend/local" + "github.com/rclone/rclone/fs" +) + +const testDialog = int64(1234567890) + +// localIndex builds an index over a temp directory using rclone's local +// backend, so BuildIndex is exercised through the same listing path a real +// remote uses rather than through a stub. +func localIndex(t *testing.T, files map[string]int) *Index { + t.Helper() + dir := t.TempDir() + for name, size := range files { + full := filepath.Join(dir, name) + if err := os.MkdirAll(filepath.Dir(full), 0o755); err != nil { + t.Fatalf("mkdir for %q: %v", name, err) + } + if err := os.WriteFile(full, make([]byte, size), 0o600); err != nil { + t.Fatalf("write %q: %v", name, err) + } + } + + ctx := context.Background() + f, err := fs.NewFs(ctx, dir) + if err != nil { + t.Fatalf("open local fs: %v", err) + } + idx, err := BuildIndex(ctx, f, testDialog) + if err != nil { + t.Fatalf("BuildIndex: %v", err) + } + return idx +} + +func TestLookupMatchesWholeNamesOnly(t *testing.T) { + idx := localIndex(t, map[string]int{ + "1234567890_4242_Pipe her! And by her, we mean pipeperr! 1080p.mp4": 4096, + "1234567890_14726_298.mp4": 2048, + }) + + if got, ok := idx.Lookup("1234567890_14726_298.mp4"); !ok || got != 2048 { + t.Errorf("Lookup(exact) = %d, %v; want 2048, true", got, ok) + } + + // The doubled '!' is the name Telegram reports; the remote holds the + // collapsed one that tdl's filenamify template wrote. Those are different + // files as far as this index is concerned, and that is the whole policy. + wanted := "1234567890_4242_Pipe her!! And by her, we mean pipeperr! 1080p.mp4" + if _, ok := idx.Lookup(wanted); ok { + t.Error("Lookup matched a near-miss name; presence must require an exact match") + } +} + +// A remote written with a subdirectory layout has to match a flat one, because +// existing archives were written both ways. +func TestBuildIndexReducesPathsToBasename(t *testing.T) { + idx := localIndex(t, map[string]int{ + "nested/dir/1234567890_42_deep.mp4": 512, + }) + if got, ok := idx.Lookup("1234567890_42_deep.mp4"); !ok || got != 512 { + t.Errorf("Lookup after basename reduction = %d, %v; want 512, true", got, ok) + } +} + +func TestStoredUnderOtherNamesFindsStaleCopies(t *testing.T) { + stale := "1234567890_4242_Pipe her! And by her, we mean pipeperr! 1080p.mp4" + idx := localIndex(t, map[string]int{stale: 4096}) + + wanted := "1234567890_4242_Pipe her!! And by her, we mean pipeperr! 1080p.mp4" + others := idx.StoredUnderOtherNames(4242, wanted) + if len(others) != 1 || others[0] != stale { + t.Fatalf("StoredUnderOtherNames = %v, want [%q]", others, stale) + } + + // The wanted name itself is never reported as an "other" name. + idx2 := localIndex(t, map[string]int{wanted: 4096}) + if others := idx2.StoredUnderOtherNames(4242, wanted); len(others) != 0 { + t.Errorf("StoredUnderOtherNames = %v, want empty when the wanted name is present", others) + } +} + +// Objects belonging to a different dialog, or not matching the stored layout at +// all, must not be indexed by id — otherwise an unrelated file could be reported +// as a stale copy of a message. +func TestStoredUnderOtherNamesIgnoresForeignObjects(t *testing.T) { + idx := localIndex(t, map[string]int{ + "999999_4242_other-dialog.mp4": 100, + "not-a-tdl-name.mp4": 100, + }) + if others := idx.StoredUnderOtherNames(4242, "1234567890_4242_x.mp4"); len(others) != 0 { + t.Errorf("StoredUnderOtherNames = %v, want empty", others) + } +} + +func TestIndexLenCountsEveryObject(t *testing.T) { + idx := localIndex(t, map[string]int{ + "1234567890_1_a.mp4": 1, + "1234567890_2_b.mp4": 1, + "unrelated.txt": 1, + }) + if idx.Len() != 3 { + t.Errorf("Len() = %d, want 3", idx.Len()) + } +} + +func TestBuildIndexOnEmptyRemote(t *testing.T) { + idx := localIndex(t, nil) + if idx.Len() != 0 { + t.Errorf("Len() = %d, want 0", idx.Len()) + } + if _, ok := idx.Lookup("anything"); ok { + t.Error("Lookup on an empty index reported a hit") + } +} + +// Two objects in different directories can share a basename. Listing order is +// not guaranteed, so silently keeping one would make the verdict flip between +// runs; the ambiguity has to be reported instead. +func TestBuildIndexReportsBasenameCollisions(t *testing.T) { + idx := localIndex(t, map[string]int{ + "a/1234567890_42_same.mp4": 100, + "b/1234567890_42_same.mp4": 0, + }) + + dup := idx.Collisions() + if len(dup) != 1 || dup[0] != "1234567890_42_same.mp4" { + t.Fatalf("Collisions() = %v, want the shared basename reported", dup) + } + if idx.Len() != 1 { + t.Errorf("Len() = %d, want 1 — a dropped collision must not be counted", idx.Len()) + } + + // The id map must not gain a duplicate entry either, or a delete would try + // the same name twice and fail the second time. + others := idx.StoredUnderOtherNames(42, "1234567890_42_wanted.mp4") + if len(others) != 1 { + t.Errorf("StoredUnderOtherNames = %v, want one entry, not a duplicate", others) + } +} + +// Acting on an object needs the path rclone knows it by, which differs from the +// basename used for matching whenever the remote has directory structure. +// Deleting by basename would miss the object, or hit the wrong one. +func TestPathOfReturnsTheFullRemotePath(t *testing.T) { + idx := localIndex(t, map[string]int{"nested/dir/1234567890_42_deep.mp4": 512}) + + const name = "1234567890_42_deep.mp4" + if _, ok := idx.Lookup(name); !ok { + t.Fatalf("Lookup(%q) missed; matching is by basename", name) + } + + got, ok := idx.PathOf(name) + if !ok { + t.Fatalf("PathOf(%q) reported no match", name) + } + if want := "nested/dir/" + name; got != want { + t.Errorf("PathOf(%q) = %q, want %q — deleting by basename would target the wrong path", name, got, want) + } +} + +func TestPathOfMissesUnknownNames(t *testing.T) { + idx := localIndex(t, nil) + if p, ok := idx.PathOf("absent.mp4"); ok { + t.Errorf("PathOf(absent) = %q, true; want no match", p) + } +} diff --git a/internal/verify/verify.go b/internal/verify/verify.go new file mode 100644 index 0000000..c72c72e --- /dev/null +++ b/internal/verify/verify.go @@ -0,0 +1,176 @@ +// Package verify answers whether a chat is fully archived on a remote. +// +// It replaces verify-export.sh and keeps that script's size judgements, which +// were arrived at by watching real failures rather than by taste. +// +// One judgement is deliberately not carried over. The script also counted files +// sitting in the staging directory as present, because download and upload were +// separate processes and a file could be finished locally but not yet uploaded +// for a whole sync interval. Here a single process owns both legs, so that state +// is not one a verify can meaningfully observe — except after an interrupted +// run, where staging may hold completed files. Phase 5 owns staging and decides +// whether to credit it; until then a verify after an interrupt may report files +// absent that are on local disk, and re-fetch them. +package verify + +import ( + "fmt" + "io" + "sort" + + "github.com/tiennm99dev/telegram-exporter/internal/naming" + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" +) + +// Index is the part of a remote snapshot verification needs. +// +// An interface rather than *remote.Index so the two questions stay separable: +// presence is answered from a whole name, and the id-keyed lookup is explicitly +// a different method with a name that says it is not an answer. It also lets the +// report be tested without a remote. +type Index interface { + // Lookup reports the size stored under an exact name. + Lookup(name string) (size int64, ok bool) + // StoredUnderOtherNames lists names present for a message id that are not + // the wanted name. Diagnostics only; never a presence answer. + StoredUnderOtherNames(messageID int, wanted string) []string +} + +// tinyThreshold is the size below which a present file is reported for a human +// to look at but still trusted. Some real media genuinely is this small, so +// treating it as damaged would re-download it forever. +const tinyThreshold = 1024 + +// Misnamed is a wanted file that is absent, while some other file is stored +// under the same message id. +type Misnamed struct { + MessageID int + Wanted string + Found []string +} + +// Unsafe is a wanted file whose name cannot be written to a path. +type Unsafe struct { + MessageID int + Name string + Reason error +} + +// Tiny is a present file small enough to be worth a look. +type Tiny struct { + MessageID int + Name string + Size int64 +} + +// Report is the outcome of comparing a chat against a remote. +type Report struct { + Expected int // media messages in the chat + Present int // present, non-empty + Bytes int64 // total size of everything expected + + Absent []int // not on the remote under the wanted name + ZeroByte []int // present but empty + + Misnamed []Misnamed + Unsafe []Unsafe + Tiny []Tiny +} + +// Todo lists the message ids needing another fetch, in ascending order. +// +// Zero-byte files are included: rclone overwrites a size-mismatched destination, +// so simply fetching again repairs them. +func (r Report) Todo() []int { + todo := make([]int, 0, len(r.Absent)+len(r.ZeroByte)) + todo = append(todo, r.Absent...) + todo = append(todo, r.ZeroByte...) + sort.Ints(todo) + return todo +} + +// Complete reports whether every expected file is present and non-empty. +func (r Report) Complete() bool { return len(r.Todo()) == 0 } + +// Check compares the wanted items against an index of the remote. +// +// Matching is on the whole name. A file stored under any other name is not the +// file that was asked for, however close it looks — that is a deliberate policy, +// and the near-misses are collected into Misnamed rather than being quietly +// accepted, because the re-download lands beside them and both copies stay. +func Check(items []tgsource.Item, idx Index) Report { + r := Report{Expected: len(items)} + + for _, it := range items { + r.Bytes += it.Size() + + if err := naming.Safe(it.Name); err != nil { + // Never counted present: this name cannot be written anywhere safe, + // so no correctly-behaving run could have archived it. + r.Unsafe = append(r.Unsafe, Unsafe{MessageID: it.MessageID, Name: it.Name, Reason: err}) + r.Absent = append(r.Absent, it.MessageID) + continue + } + + size, ok := idx.Lookup(it.Name) + switch { + case !ok: + r.Absent = append(r.Absent, it.MessageID) + if others := idx.StoredUnderOtherNames(it.MessageID, it.Name); len(others) > 0 { + r.Misnamed = append(r.Misnamed, Misnamed{ + MessageID: it.MessageID, Wanted: it.Name, Found: others, + }) + } + case size == 0: + r.ZeroByte = append(r.ZeroByte, it.MessageID) + default: + r.Present++ + if size < tinyThreshold { + r.Tiny = append(r.Tiny, Tiny{MessageID: it.MessageID, Name: it.Name, Size: size}) + } + } + } + return r +} + +// Write renders a report in the shape verify-export.sh printed, so the numbers +// stay comparable across the cutover. +func (r Report) Write(w io.Writer) { + fmt.Fprintf(w, "media expected : %d (%.1f GiB)\n", r.Expected, float64(r.Bytes)/(1<<30)) + fmt.Fprintf(w, "present and intact : %d\n", r.Present) + fmt.Fprintf(w, " absent : %d\n", len(r.Absent)) + fmt.Fprintf(w, " zero-byte : %d\n", len(r.ZeroByte)) + + if len(r.Tiny) > 0 { + fmt.Fprintf(w, " under 1KiB (check, not retried): %d\n", len(r.Tiny)) + for _, t := range r.Tiny[:min(5, len(r.Tiny))] { + fmt.Fprintf(w, " id %d %d B %q\n", t.MessageID, t.Size, t.Name) + } + } + + if len(r.Unsafe) > 0 { + fmt.Fprintf(w, "\nunsafe filenames : %d\n", len(r.Unsafe)) + fmt.Fprintf(w, " these cannot be written to a path and are never fetched:\n") + for _, u := range r.Unsafe { + fmt.Fprintf(w, " id %d %v\n", u.MessageID, u.Reason) + } + } + + if len(r.Misnamed) > 0 { + fmt.Fprintf(w, "\nstored under a different name : %d\n", len(r.Misnamed)) + fmt.Fprintf(w, " counted as absent and fetched again; delete the stale copies so the\n") + fmt.Fprintf(w, " re-download does not leave two files for the same message:\n") + for _, m := range r.Misnamed { + fmt.Fprintf(w, " id %d\n wanted: %q\n", m.MessageID, m.Wanted) + for _, f := range m.Found { + fmt.Fprintf(w, " remote: %q\n", f) + } + } + } + + if todo := r.Todo(); len(todo) > 0 { + fmt.Fprintf(w, "\nneeds another pass : %d (ids %d–%d)\n", len(todo), todo[0], todo[len(todo)-1]) + return + } + fmt.Fprintf(w, "\nCOMPLETE: every media message is present and non-empty.\n") +} diff --git a/internal/verify/verify_test.go b/internal/verify/verify_test.go new file mode 100644 index 0000000..6978b77 --- /dev/null +++ b/internal/verify/verify_test.go @@ -0,0 +1,181 @@ +package verify + +import ( + "slices" + "strings" + "testing" + + "github.com/iyear/tdl/core/tmedia" + + "github.com/tiennm99dev/telegram-exporter/internal/naming" + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" +) + +const dialog = int64(1234567890) + +// fakeIndex is a remote snapshot expressed directly, so report logic is tested +// without a network or a filesystem. +type fakeIndex map[string]int64 + +func (f fakeIndex) Lookup(name string) (int64, bool) { + size, ok := f[name] + return size, ok +} + +func (f fakeIndex) StoredUnderOtherNames(messageID int, wanted string) []string { + var others []string + for name := range f { + if name == wanted { + continue + } + if id, ok := naming.SplitStored(dialog, name); ok && id == messageID { + others = append(others, name) + } + } + slices.Sort(others) + return others +} + +func item(msgID int, file string, size int64) tgsource.Item { + m := &tmedia.Media{Name: file, Size: size} + return tgsource.Item{ + DialogID: dialog, + MessageID: msgID, + Name: naming.For(dialog, msgID, m), + Media: m, + } +} + +func TestCheckClassifiesEveryOutcome(t *testing.T) { + items := []tgsource.Item{ + item(1, "present.mp4", 5000), + item(2, "empty.mp4", 5000), + item(3, "absent.mp4", 5000), + item(4, "tiny.jpg", 500), + } + idx := fakeIndex{ + "1234567890_1_present.mp4": 5000, + "1234567890_2_empty.mp4": 0, + "1234567890_4_tiny.jpg": 500, + } + + r := Check(items, idx) + + if r.Expected != 4 { + t.Errorf("Expected = %d, want 4", r.Expected) + } + if r.Present != 2 { + t.Errorf("Present = %d, want 2 (the non-empty ones)", r.Present) + } + if !slices.Equal(r.Absent, []int{3}) { + t.Errorf("Absent = %v, want [3]", r.Absent) + } + if !slices.Equal(r.ZeroByte, []int{2}) { + t.Errorf("ZeroByte = %v, want [2]", r.ZeroByte) + } + + // Sub-1KiB is reported but still counted present: some real media is + // genuinely that small, so retrying it would loop forever. + if len(r.Tiny) != 1 || r.Tiny[0].MessageID != 4 { + t.Errorf("Tiny = %v, want just message 4", r.Tiny) + } + if slices.Contains(r.Todo(), 4) { + t.Error("a tiny file must not be queued for another fetch") + } + + // Zero-byte files are retried: rclone overwrites a size-mismatched + // destination, so fetching again repairs them. + if want := []int{2, 3}; !slices.Equal(r.Todo(), want) { + t.Errorf("Todo() = %v, want %v", r.Todo(), want) + } + if r.Complete() { + t.Error("Complete() = true with work outstanding") + } +} + +func TestCheckCompleteWhenEverythingIsPresent(t *testing.T) { + items := []tgsource.Item{item(1, "a.mp4", 10), item(2, "b.mp4", 20)} + idx := fakeIndex{"1234567890_1_a.mp4": 10, "1234567890_2_b.mp4": 20} + + r := Check(items, idx) + if !r.Complete() { + t.Fatalf("Complete() = false, Todo() = %v", r.Todo()) + } + var sb strings.Builder + r.Write(&sb) + if !strings.Contains(sb.String(), "COMPLETE") { + t.Errorf("report should say COMPLETE, got:\n%s", sb.String()) + } +} + +// The message that motivated the rewrite: the wanted name has a doubled '!', +// the remote holds the collapsed one that tdl's template wrote. It must be +// absent, and the stale copy must be surfaced rather than silently accepted. +func TestCheckReportsMisnamedCopiesAsAbsent(t *testing.T) { + items := []tgsource.Item{item(4242, "Pipe her!! And by her, we mean pipeperr! 1080p.mp4", 966444937)} + stale := "1234567890_4242_Pipe her! And by her, we mean pipeperr! 1080p.mp4" + idx := fakeIndex{stale: 966444937} + + r := Check(items, idx) + + if !slices.Equal(r.Absent, []int{4242}) { + t.Errorf("Absent = %v, want [4242] — a near-miss name is a different file", r.Absent) + } + if r.Present != 0 { + t.Errorf("Present = %d, want 0", r.Present) + } + if len(r.Misnamed) != 1 || !slices.Equal(r.Misnamed[0].Found, []string{stale}) { + t.Fatalf("Misnamed = %+v, want the stale copy reported", r.Misnamed) + } + + var sb strings.Builder + r.Write(&sb) + out := sb.String() + for _, want := range []string{"stored under a different name", stale, "delete the stale copies"} { + if !strings.Contains(out, want) { + t.Errorf("report should mention %q, got:\n%s", want, out) + } + } +} + +// A name that cannot be written to a path is never counted present and never +// silently skipped — it is reported and queued, so it stays visible. +func TestCheckFlagsUnsafeNames(t *testing.T) { + items := []tgsource.Item{item(7, "../../../.config/rclone/rclone.conf", 100)} + + r := Check(items, fakeIndex{}) + + if len(r.Unsafe) != 1 || r.Unsafe[0].MessageID != 7 { + t.Fatalf("Unsafe = %+v, want message 7 flagged", r.Unsafe) + } + if !slices.Equal(r.Absent, []int{7}) { + t.Errorf("Absent = %v, want [7]", r.Absent) + } + if r.Present != 0 { + t.Errorf("Present = %d, want 0", r.Present) + } + + var sb strings.Builder + r.Write(&sb) + if !strings.Contains(sb.String(), "unsafe filenames") { + t.Errorf("report should flag the unsafe name, got:\n%s", sb.String()) + } +} + +// An unsafe name must be rejected even when something is stored under that +// message id: the index is not the authority on whether a name is writable. +func TestCheckUnsafeNameIsNeverPresent(t *testing.T) { + items := []tgsource.Item{item(7, "sub/dir.mp4", 100)} + idx := fakeIndex{"1234567890_7_sub/dir.mp4": 100} + + if r := Check(items, idx); r.Present != 0 || len(r.Unsafe) != 1 { + t.Errorf("Present = %d, Unsafe = %+v; an unsafe name must never count present", r.Present, r.Unsafe) + } +} + +func TestReportTodoIsSorted(t *testing.T) { + r := Report{Absent: []int{4242, 3}, ZeroByte: []int{100}} + if want := []int{3, 100, 4242}; !slices.Equal(r.Todo(), want) { + t.Errorf("Todo() = %v, want %v", r.Todo(), want) + } +} From a81e7eaacbdacda299f72114fd7927d3c92cbd2e Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Sun, 6 Sep 2026 19:21:46 +0700 Subject: [PATCH 04/16] feat: download, upload and drive a chat to completion in one process MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Phases 4 through 6: the two legs and the command that joins them. Downloads go to .part and are renamed only once complete, so a file without the suffix is always whole. That is what lets the upload leg treat "exists" as "finished" — the property run.sh could only approximate with a filename convention plus an age guard, because it could not see inside tdl. Every finished file is checked against the size Telegram reported, and that check rather than the error is the authoritative signal. core's downloader logs a failed transfer and returns nil, and its completion callback is deferred on that named return, so a failure arrives indistinguishable from a success. Trusting it would promote a truncated file and archive it as complete. The disk cap is a semaphore over bytes. A download reserves its own size before starting and releases it only after the upload confirms, so a slow remote stalls downloads by itself. Blocking the iterator is safe because the downloader calls it from its dispatch loop while workers run in a group, so a blocked iterator never stops the uploads that free the space. Gone with it: the du polling, the SIGSTOP and SIGCONT suspension, the min-age guard, the temp-file filter and the sweep-failure counter. A cap smaller than the largest file is refused up front. The semaphore could never admit it, and a run blocked on a file it can never start looks exactly like a stalled remote. Uploads re-state each object to prove its size before the local copy is gone, closing a gap where a truncated upload was only noticed by a later verify. The destination is created before the chat is read. It is also the credentials check, and doing it first means a bad destination fails in seconds rather than after a full history walk. One invocation converges: each item is checked against the index immediately before download, so there are no passes and re-running is the resume path. Options that no longer exist say what replaced them instead of failing as unknown flags. Verified end to end against the live chat and a scratch remote path: two files downloaded, uploaded, confirmed present at the right size, staging left empty. --- cmd/tgexport/main.go | 5 +- cmd/tgexport/sync.go | 322 +++++++++++++++++++++++++++++ cmd/tgexport/sync_test.go | 139 +++++++++++++ internal/pipeline/download.go | 172 +++++++++++++++ internal/pipeline/download_test.go | 290 ++++++++++++++++++++++++++ internal/pipeline/elem.go | 13 ++ internal/pipeline/pipeline.go | 193 +++++++++++++++++ internal/pipeline/pipeline_test.go | 147 +++++++++++++ internal/pipeline/progress.go | 105 ++++++++++ internal/pipeline/upload.go | 47 +++++ internal/remote/fs.go | 14 ++ internal/remote/index.go | 7 + internal/report/progress.go | 108 ++++++++++ 13 files changed, 1561 insertions(+), 1 deletion(-) create mode 100644 cmd/tgexport/sync.go create mode 100644 cmd/tgexport/sync_test.go create mode 100644 internal/pipeline/download.go create mode 100644 internal/pipeline/download_test.go create mode 100644 internal/pipeline/pipeline.go create mode 100644 internal/pipeline/pipeline_test.go create mode 100644 internal/pipeline/progress.go create mode 100644 internal/pipeline/upload.go create mode 100644 internal/report/progress.go diff --git a/cmd/tgexport/main.go b/cmd/tgexport/main.go index 0a515fe..7314d8f 100644 --- a/cmd/tgexport/main.go +++ b/cmd/tgexport/main.go @@ -65,6 +65,8 @@ func run() int { err = listCmd(ctx, os.Args[2:]) case "verify": err = verifyCmd(ctx, os.Args[2:]) + case "sync": + err = syncCmd(ctx, os.Args[2:]) default: fmt.Fprintf(os.Stderr, "unknown command %q\n\n", os.Args[1]) usage() @@ -150,9 +152,10 @@ func usage() { fmt.Fprint(os.Stderr, `Usage: tgexport [options] Commands: - doctor Check the Telegram session, the destination remote, and free space + sync Archive a chat to a remote, fetching only what is missing list Print every media message in a chat as idsizename verify Report whether a chat is fully archived on a remote + doctor Check the Telegram session, the destination remote, and free space Run 'tgexport -h' for command options. `) diff --git a/cmd/tgexport/sync.go b/cmd/tgexport/sync.go new file mode 100644 index 0000000..207b139 --- /dev/null +++ b/cmd/tgexport/sync.go @@ -0,0 +1,322 @@ +package main + +import ( + "context" + "errors" + "flag" + "fmt" + "iter" + "os" + + "github.com/iyear/tdl/core/dcpool" + tdlstorage "github.com/iyear/tdl/core/storage" + "github.com/rclone/rclone/fs" + + "github.com/tiennm99dev/telegram-exporter/internal/pipeline" + "github.com/tiennm99dev/telegram-exporter/internal/remote" + "github.com/tiennm99dev/telegram-exporter/internal/report" + "github.com/tiennm99dev/telegram-exporter/internal/tdlkv" + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" + "github.com/tiennm99dev/telegram-exporter/internal/verify" +) + +// retiredFlags map options the shell pipeline had onto what replaced them. +// +// Recognising them beats "flag provided but not defined": these were in +// muscle memory and in wrapper scripts, and a bare parse error does not say +// whether the concept moved or disappeared. +var retiredFlags = map[string]string{ + "i": "the rclone sweep interval is gone; uploads start the moment a download finishes", + "a": "--min-age is gone; a file is only uploaded once the downloader reports it complete", + "f": "the export JSON is gone; the chat is read live, so names cannot go stale", + "p": "there are no passes; one invocation converges, and re-running resumes", + "q": "renamed to --min-free", +} + +// syncCmd archives a chat to a remote: read the chat, skip what is already +// there, download and upload the rest, then report on the result. +func syncCmd(ctx context.Context, args []string) error { + fs := flag.NewFlagSet("sync", flag.ContinueOnError) + var ( + chat = fs.String("c", "", "chat id, username, or t.me link (required)") + remoteArg = fs.String("r", "", "rclone destination, e.g. pikpak:archive (required)") + staging = fs.String("d", "./staging", "staging directory for files in flight") + maxStaging = fs.String("m", "", "cap staging at this size, e.g. 40G (default: no cap)") + threads = fs.Int("threads", 4, "connections per file") + limit = fs.Int("limit", 2, "files downloading at once") + uploads = fs.Int("uploads", 2, "files uploading at once") + minFree = fs.Int64("min-free", 5, "stop if the remote has fewer than this many GiB free") + limitItems = fs.Int("limit-items", 0, "stop after this many files (0 means no limit)") + confirm = fs.Bool("confirm", true, "re-state each uploaded file to prove its size") + takeout = fs.Bool("takeout", true, "use a takeout session, as `tdl dl --takeout` did") + ns = fs.String("n", "default", "tdl session namespace") + dataDir = fs.String("storage", tdlkv.DefaultDir(), "tdl bolt storage directory") + ) + for name, replacement := range retiredFlags { + fs.Var(retiredFlag{name, replacement}, name, "retired") + } + + if err := fs.Parse(args); err != nil { + if errors.Is(err, flag.ErrHelp) { + return err + } + return fmt.Errorf("%w: %v", errUsage, err) + } + if *chat == "" || *remoteArg == "" { + return fmt.Errorf("%w: -c CHAT and -r REMOTE:PATH are both required", errUsage) + } + + budget, err := parseSize(*maxStaging) + if err != nil { + return fmt.Errorf("%w: -m %v", errUsage, err) + } + + ctx, err = remote.Init(ctx, remote.DefaultTunables()) + if err != nil { + return err + } + dst, err := remote.Resolve(ctx, *remoteArg) + if err != nil { + return err + } + + // Partial files from an earlier run cannot be continued — core's downloader + // takes no starting offset — so they are cleared before anything else fills + // the disk with fragments no run will finish. + if swept, err := pipeline.SweepPartials(*staging); err != nil { + return fmt.Errorf("clear partial downloads: %w", err) + } else if swept > 0 { + fmt.Fprintf(os.Stderr, "cleared %d partial download(s) from an earlier run\n", swept) + } + + // Before anything expensive: prove the destination is reachable and writable. + if err := remote.EnsureDir(ctx, dst); err != nil { + return err + } + if err := checkFree(ctx, dst, *minFree); err != nil { + return err + } + + kv, err := tdlkv.Open(*dataDir, *ns) + if err != nil { + return err + } + defer func() { _ = kv.Close() }() + + sess, err := tgsource.New(ctx, tgsource.Options{KV: kv}) + if err != nil { + return err + } + + var final verify.Report + if err := sess.Run(ctx, func(ctx context.Context, pool dcpool.Pool) error { + api := pool.Default(ctx) + + peer, err := tgsource.ResolveChat(ctx, tgsource.Manager(api, tdlstorage.NewPeers(kv)), *chat) + if err != nil { + return err + } + + fmt.Fprintf(os.Stderr, "reading %s\n", *chat) + var items []tgsource.Item + for it, err := range tgsource.Walk(ctx, api, peer) { + if err != nil { + return err + } + items = append(items, it) + } + + idx, err := remote.BuildIndex(ctx, dst, peer.ID()) + if err != nil { + return err + } + before := verify.Check(items, idx) + fmt.Fprintf(os.Stderr, "%d media messages, %d already archived, %d to fetch\n", + before.Expected, before.Present, len(before.Todo())) + + todo := selectTodo(items, before, *limitItems) + if len(todo) == 0 { + final = before + return nil + } + + if err := validateBudget(budget, todo); err != nil { + return err + } + + var todoBytes int64 + for _, it := range todo { + todoBytes += it.Size() + } + fmt.Fprintf(os.Stderr, "fetching %d file(s), %.1f GiB\n", len(todo), float64(todoBytes)/(1<<30)) + + rep := report.New(os.Stderr, len(todo), todoBytes) + res, runErr := pipeline.Run(ctx, sliceSeq(todo), pipeline.Options{ + Pool: pool, + Dst: dst, + Staging: *staging, + Threads: *threads, + Limit: *limit, + Uploads: *uploads, + Budget: budget, + Confirm: *confirm, + Takeout: *takeout, + Report: rep.Update, + }) + rep.Finish(res.Stats) + + for _, f := range res.Failed() { + fmt.Fprintf(os.Stderr, " message %d failed: %v\n", f.Item.MessageID, f.Err) + } + if runErr != nil { + return runErr + } + + // The remote is re-indexed rather than assumed: the run's own view of + // what it uploaded is exactly the thing under test. + idx, err = remote.BuildIndex(ctx, dst, peer.ID()) + if err != nil { + return err + } + final = verify.Check(items, idx) + return nil + }); err != nil { + return err + } + + fmt.Fprintln(os.Stderr) + final.Write(os.Stdout) + + if !final.Complete() { + return fmt.Errorf("%w: %d file(s) still to fetch", errIncomplete, len(final.Todo())) + } + return nil +} + +// selectTodo picks the items still needing a fetch, newest first, optionally +// capped for a smoke test. +func selectTodo(items []tgsource.Item, r verify.Report, limit int) []tgsource.Item { + want := make(map[int]struct{}, len(r.Todo())) + for _, id := range r.Todo() { + want[id] = struct{}{} + } + // Unsafe names are in Todo so they stay visible in the report, but fetching + // one is impossible by definition, so it is not queued for download. + for _, u := range r.Unsafe { + delete(want, u.MessageID) + } + + var todo []tgsource.Item + for _, it := range items { + if _, ok := want[it.MessageID]; !ok { + continue + } + todo = append(todo, it) + if limit > 0 && len(todo) == limit { + break + } + } + return todo +} + +func sliceSeq(items []tgsource.Item) iter.Seq2[tgsource.Item, error] { + return func(yield func(tgsource.Item, error) bool) { + for _, it := range items { + if !yield(it, nil) { + return + } + } + } +} + +// validateBudget refuses a cap smaller than the largest file. +// +// The semaphore can never admit a weight above its limit, so such a run would +// block forever on a file it could never start — indistinguishable, from the +// outside, from a stalled remote. run.sh could only warn about this after the +// fact, once draining failed to get back under the cap. +func validateBudget(budget int64, todo []tgsource.Item) error { + if budget <= 0 { + return nil + } + var largest int64 + for _, it := range todo { + if it.Size() > largest { + largest = it.Size() + } + } + if largest > budget { + return fmt.Errorf("%w: -m is %.1f GiB but the largest file to fetch is %.1f GiB; "+ + "the cap must exceed the biggest single file", + errUsage, float64(budget)/(1<<30), float64(largest)/(1<<30)) + } + return nil +} + +// checkFree refuses to start when the remote is nearly full. +// +// A backend that cannot report a quota is treated as unlimited rather than as a +// failure — the shell pipeline made that choice deliberately so a remote without +// an About API never blocked a run, and it is preserved. +func checkFree(ctx context.Context, dst fs.Fs, minGiB int64) error { + free, ok := remote.FreeBytes(ctx, dst) + if !ok { + return nil + } + freeGiB := free / (1 << 30) + fmt.Fprintf(os.Stderr, "%s has %d GiB free\n", dst.String(), freeGiB) + if freeGiB < minGiB { + return fmt.Errorf("%s has only %d GiB free, below the %d GiB floor; "+ + "free space or lower --min-free", dst.String(), freeGiB, minGiB) + } + return nil +} + +// retiredFlag reports a helpful error for an option that no longer exists. +type retiredFlag struct{ name, replacement string } + +func (r retiredFlag) String() string { return "" } +func (r retiredFlag) Set(string) error { + return fmt.Errorf("-%s no longer exists: %s", r.name, r.replacement) +} + +// parseSize reads a binary size such as 40G, matching what run.sh -m accepted. +func parseSize(s string) (int64, error) { + if s == "" { + return 0, nil + } + mult := int64(1) + switch unit := s[len(s)-1]; unit { + case 'K', 'k': + mult = 1 << 10 + case 'M', 'm': + mult = 1 << 20 + case 'G', 'g': + mult = 1 << 30 + case 'T', 't': + mult = 1 << 40 + default: + if unit < '0' || unit > '9' { + return 0, fmt.Errorf("unknown size suffix %q, expected K, M, G or T", string(unit)) + } + } + digits := s + if mult > 1 { + digits = s[:len(s)-1] + } + + var n int64 + if digits == "" { + return 0, fmt.Errorf("%q has no number", s) + } + for _, r := range digits { + if r < '0' || r > '9' { + return 0, fmt.Errorf("%q is not a size", s) + } + n = n*10 + int64(r-'0') + } + if n <= 0 { + return 0, fmt.Errorf("must be greater than zero") + } + return n * mult, nil +} diff --git a/cmd/tgexport/sync_test.go b/cmd/tgexport/sync_test.go new file mode 100644 index 0000000..9cd6d10 --- /dev/null +++ b/cmd/tgexport/sync_test.go @@ -0,0 +1,139 @@ +package main + +import ( + "errors" + "strings" + "testing" + + "github.com/iyear/tdl/core/tmedia" + + "github.com/tiennm99dev/telegram-exporter/internal/naming" + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" + "github.com/tiennm99dev/telegram-exporter/internal/verify" +) + +func TestParseSize(t *testing.T) { + ok := map[string]int64{ + "": 0, // unset means no cap + "40G": 40 << 30, + "40g": 40 << 30, + "512M": 512 << 20, + "2T": 2 << 40, + "1024": 1024, // bare number is bytes + "1K": 1 << 10, + } + for in, want := range ok { + got, err := parseSize(in) + if err != nil { + t.Errorf("parseSize(%q) = %v, want %d", in, err, want) + continue + } + if got != want { + t.Errorf("parseSize(%q) = %d, want %d", in, got, want) + } + } + + for _, in := range []string{"G", "0", "0G", "-5G", "40GB", "4.5G", "abc", "40Q"} { + if got, err := parseSize(in); err == nil { + t.Errorf("parseSize(%q) = %d, want an error", in, got) + } + } +} + +func syncItem(id int, size int64) tgsource.Item { + m := &tmedia.Media{Name: "f.mp4", Size: size} + return tgsource.Item{DialogID: 1, MessageID: id, Name: naming.For(1, id, m), Media: m} +} + +func TestSelectTodoPicksOnlyOutstandingItems(t *testing.T) { + items := []tgsource.Item{syncItem(1, 10), syncItem(2, 10), syncItem(3, 10), syncItem(4, 10)} + r := verify.Report{Absent: []int{2, 4}, ZeroByte: []int{3}} + + got := selectTodo(items, r, 0) + if len(got) != 3 { + t.Fatalf("selected %d items, want 3", len(got)) + } + for _, it := range got { + if it.MessageID == 1 { + t.Error("selected an item that is already archived") + } + } +} + +// An unsafe name is kept in the report so it stays visible, but queuing it for +// download would retry something that can never succeed — the non-terminating +// loop this design exists to avoid. +func TestSelectTodoSkipsUnsafeNames(t *testing.T) { + items := []tgsource.Item{syncItem(1, 10), syncItem(2, 10)} + r := verify.Report{ + Absent: []int{1, 2}, + Unsafe: []verify.Unsafe{{MessageID: 2, Name: "../x", Reason: errors.New("unsafe")}}, + } + + got := selectTodo(items, r, 0) + if len(got) != 1 || got[0].MessageID != 1 { + t.Errorf("selected %+v, want only message 1", got) + } +} + +func TestSelectTodoHonoursLimit(t *testing.T) { + items := []tgsource.Item{syncItem(1, 10), syncItem(2, 10), syncItem(3, 10)} + r := verify.Report{Absent: []int{1, 2, 3}} + + if got := selectTodo(items, r, 2); len(got) != 2 { + t.Errorf("selected %d items with a limit of 2, want 2", len(got)) + } + if got := selectTodo(items, r, 0); len(got) != 3 { + t.Errorf("selected %d items with no limit, want 3", len(got)) + } +} + +// A cap below the largest file could never admit it, so the run would block on +// something it can never start — which from outside looks like a stalled remote. +// It has to be refused up front. +func TestValidateBudgetRejectsCapBelowLargestFile(t *testing.T) { + todo := []tgsource.Item{syncItem(1, 1<<20), syncItem(2, 5<<30)} + + err := validateBudget(4<<30, todo) + if err == nil { + t.Fatal("a 4 GiB cap was accepted with a 5 GiB file to fetch") + } + if !errors.Is(err, errUsage) { + t.Errorf("error should be a usage error, got: %v", err) + } + if !strings.Contains(err.Error(), "largest file") { + t.Errorf("error should name the problem, got: %v", err) + } + + if err := validateBudget(6<<30, todo); err != nil { + t.Errorf("a 6 GiB cap should accept a 5 GiB file, got: %v", err) + } + if err := validateBudget(0, todo); err != nil { + t.Errorf("an unset cap should accept anything, got: %v", err) + } +} + +// Options the shell pipeline had must produce an explanation, not "flag +// provided but not defined" — they are in wrapper scripts and muscle memory. +func TestRetiredFlagsExplainWhatReplacedThem(t *testing.T) { + for name, replacement := range retiredFlags { + err := retiredFlag{name, replacement}.Set("x") + if err == nil { + t.Errorf("-%s was accepted, want an explanation", name) + continue + } + if !strings.Contains(err.Error(), "-"+name) { + t.Errorf("error for -%s should name the flag, got: %v", name, err) + } + if !strings.Contains(err.Error(), replacement) { + t.Errorf("error for -%s should say what replaced it, got: %v", name, err) + } + } + + // The ones that mattered most in run.sh. + for _, name := range []string{"i", "a", "f", "p", "q"} { + if _, ok := retiredFlags[name]; !ok { + t.Errorf("-%s was a run.sh flag but is not recognised as retired", name) + } + } +} diff --git a/internal/pipeline/download.go b/internal/pipeline/download.go new file mode 100644 index 0000000..e160c0a --- /dev/null +++ b/internal/pipeline/download.go @@ -0,0 +1,172 @@ +package pipeline + +import ( + "context" + "errors" + "fmt" + "iter" + "os" + "path/filepath" + "strings" + + "github.com/iyear/tdl/core/dcpool" + "github.com/iyear/tdl/core/downloader" + + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" +) + +// DownloadOptions configures a download run. +type DownloadOptions struct { + Pool dcpool.Pool + Staging string + Threads int // connections per file + Limit int // files in flight + Takeout bool // use a takeout session, as `tdl dl --takeout` does + + // Report, when set, is called with a running summary. It is invoked from + // download worker goroutines, so it must be cheap and safe to call + // concurrently. + Report func(Stats) + + // acquire reserves staging space before a download starts, blocking until + // there is room. Unset means no bound. + acquire func(context.Context, int64) error + // onReady hands a completed file to the upload leg; onFailed says nothing + // was staged, so whatever acquire reserved must be given back. + onReady func(tgsource.Item) + onFailed func(tgsource.Item) +} + +// Download fetches every item in seq into the staging directory. +// +// Each file is written to .part and renamed to only once the +// downloader reports it complete, so a name without the suffix is always a +// whole file. That is what lets the upload half treat "the file exists" as +// "the file is finished" — the property the shell pipeline had to approximate +// with a filename convention plus an age guard, because it could not see +// inside tdl. +// +// A failed item does not abort the run: it is recorded in the returned outcomes +// and the rest continue, matching what a partial `tdl dl` pass did. +func Download(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o DownloadOptions) ([]Outcome, Stats, error) { + if o.Threads <= 0 { + o.Threads = 4 + } + if o.Limit <= 0 { + o.Limit = 2 + } + if err := os.MkdirAll(o.Staging, 0o755); err != nil { + return nil, Stats{}, fmt.Errorf("create staging directory: %w", err) + } + + it := newElemIter(seq, o.Staging, o.Takeout) + it.acquire = o.acquire + defer func() { _ = it.Close() }() + + prog := newProgress(func(e *elem, err error) error { + ferr := finish(o.Staging, e, err) + if err == nil && ferr == nil { + if o.onReady != nil { + o.onReady(e.item) + } + return nil + } + if o.onFailed != nil { + o.onFailed(e.item) + } + return ferr + }, o.Report) + + err := downloader.New(downloader.Options{ + Pool: o.Pool, + Threads: o.Threads, + Iter: it, + Progress: prog, + }).Download(ctx, o.Limit) + + outcomes, stats := prog.results() + return outcomes, stats, err +} + +// finish closes a downloaded file and either promotes it or removes it. +// +// The size on disk is checked against the size Telegram reported, and that check +// is not belt-and-braces — it is the only reliable failure signal available. +// core's Download swallows non-cancellation errors: it logs them and returns +// nil, and OnDone is deferred on that named return, so a failed transfer arrives +// here indistinguishable from a successful one (downloader.go:47-60). Trusting +// the error alone would promote a truncated file to its final name, and the +// upload leg would archive it as complete. +// +// A partial file is deleted rather than kept: the downloader exposes no resume +// offset, so a leftover .part could never be continued, and leaving one behind +// would only invite a later run to mistake it for progress. +func finish(staging string, e *elem, downloadErr error) error { + part := partPath(staging, e.item) + + if cerr := e.file.Close(); cerr != nil && downloadErr == nil { + downloadErr = cerr + } + + if downloadErr == nil { + if err := checkSize(part, e.item.Size()); err != nil { + downloadErr = err + } + } + + if downloadErr != nil { + if rerr := os.Remove(part); rerr != nil && !os.IsNotExist(rerr) { + return fmt.Errorf("remove partial %q: %w", part, rerr) + } + // Returned so the caller records a failure even when the downloader + // claimed success; otherwise a short file would vanish silently and the + // run would report itself complete. + return downloadErr + } + + if err := os.Rename(part, finalPath(staging, e.item)); err != nil { + return fmt.Errorf("promote %q: %w", part, err) + } + return nil +} + +// checkSize compares what landed on disk against what Telegram said the file is. +func checkSize(path string, want int64) error { + info, err := os.Stat(path) + if err != nil { + return fmt.Errorf("stat downloaded file: %w", err) + } + if info.Size() != want { + return fmt.Errorf("short download: got %d bytes, expected %d", info.Size(), want) + } + return nil +} + +// SweepPartials removes leftover .part files from an earlier run. +// +// They cannot be resumed — core's downloader takes no starting offset — so the +// only options are delete or accumulate, and accumulating fills the disk with +// fragments no run will ever finish. +func SweepPartials(staging string) (int, error) { + entries, err := os.ReadDir(staging) + if err != nil { + if os.IsNotExist(err) { + return 0, nil + } + return 0, fmt.Errorf("read staging directory: %w", err) + } + + removed := 0 + var errs []error + for _, entry := range entries { + if entry.IsDir() || !strings.HasSuffix(entry.Name(), partSuffix) { + continue + } + if err := os.Remove(filepath.Join(staging, entry.Name())); err != nil { + errs = append(errs, err) + continue + } + removed++ + } + return removed, errors.Join(errs...) +} diff --git a/internal/pipeline/download_test.go b/internal/pipeline/download_test.go new file mode 100644 index 0000000..7940c46 --- /dev/null +++ b/internal/pipeline/download_test.go @@ -0,0 +1,290 @@ +package pipeline + +import ( + "errors" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/gotd/td/tg" + "github.com/iyear/tdl/core/downloader" + "github.com/iyear/tdl/core/tmedia" + + "github.com/tiennm99dev/telegram-exporter/internal/naming" + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" +) + +func testItem(t *testing.T, msgID int, file string, size int64) tgsource.Item { + t.Helper() + m := &tmedia.Media{ + Name: file, + Size: size, + DC: 2, + InputFileLoc: &tg.InputDocumentFileLocation{ID: int64(msgID)}, + } + return tgsource.Item{ + DialogID: 1234567890, + MessageID: msgID, + Name: naming.For(1234567890, msgID, m), + Media: m, + } +} + +// openElem mimics what elemIter does, so finish can be tested without a network. +func openElem(t *testing.T, staging string, it tgsource.Item) *elem { + t.Helper() + f, err := os.OpenFile(partPath(staging, it), os.O_CREATE|os.O_RDWR, 0o600) + if err != nil { + t.Fatalf("open part file: %v", err) + } + return &elem{item: it, file: f} +} + +// A name without the suffix must always be a whole file: that is the property +// the upload half relies on to treat "exists" as "finished", replacing the +// filename-convention-plus-age-guard the shell pipeline needed. +func TestFinishPromotesOnlyOnSuccess(t *testing.T) { + staging := t.TempDir() + it := testItem(t, 1, "video.mp4", 100) + + e := openElem(t, staging, it) + if _, err := e.file.WriteAt(make([]byte, it.Size()), 0); err != nil { + t.Fatalf("write: %v", err) + } + if err := finish(staging, e, nil); err != nil { + t.Fatalf("finish: %v", err) + } + + if _, err := os.Stat(finalPath(staging, it)); err != nil { + t.Errorf("final file missing after a successful download: %v", err) + } + if _, err := os.Stat(partPath(staging, it)); !os.IsNotExist(err) { + t.Errorf("part file still present after promotion") + } +} + +func TestFinishRemovesPartialOnFailure(t *testing.T) { + staging := t.TempDir() + it := testItem(t, 2, "video.mp4", 100) + + e := openElem(t, staging, it) + if _, err := e.file.WriteAt([]byte("half"), 0); err != nil { + t.Fatalf("write: %v", err) + } + // finish returns the failure rather than swallowing it, so the caller + // records the item as failed instead of quietly counting it done. + want := errors.New("connection reset") + if err := finish(staging, e, want); !errors.Is(err, want) { + t.Fatalf("finish = %v, want the download error returned", err) + } + + // Neither file may survive: a partial promoted to the final name would be + // indistinguishable from a complete download and would never be repaired. + if _, err := os.Stat(partPath(staging, it)); !os.IsNotExist(err) { + t.Errorf("part file survived a failed download") + } + if _, err := os.Stat(finalPath(staging, it)); !os.IsNotExist(err) { + t.Errorf("a failed download was promoted to the final name") + } +} + +func TestSweepPartialsRemovesOnlyPartFiles(t *testing.T) { + staging := t.TempDir() + keep := filepath.Join(staging, "1234567890_1_done.mp4") + drop := filepath.Join(staging, "1234567890_2_wip.mp4"+partSuffix) + legacy := filepath.Join(staging, "1234567890_3_old.mp4.tmp") + + for _, p := range []string{keep, drop, legacy} { + if err := os.WriteFile(p, []byte("x"), 0o600); err != nil { + t.Fatalf("seed %q: %v", p, err) + } + } + + removed, err := SweepPartials(staging) + if err != nil { + t.Fatalf("SweepPartials: %v", err) + } + if removed != 1 { + t.Errorf("removed = %d, want 1", removed) + } + if _, err := os.Stat(keep); err != nil { + t.Errorf("a completed file was swept: %v", err) + } + // tdl's own suffix is left alone: a shared staging directory during the + // cutover may hold files a legacy run is still writing. + if _, err := os.Stat(legacy); err != nil { + t.Errorf("a legacy tdl .tmp file was swept: %v", err) + } +} + +func TestSweepPartialsOnMissingDirectory(t *testing.T) { + removed, err := SweepPartials(filepath.Join(t.TempDir(), "absent")) + if err != nil { + t.Errorf("SweepPartials on a missing directory = %v, want nil", err) + } + if removed != 0 { + t.Errorf("removed = %d, want 0", removed) + } +} + +// An unwritable name must stop the iterator with a message naming the message, +// rather than surfacing as a bare os.Create failure later. +func TestElemIterRejectsUnsafeNames(t *testing.T) { + staging := t.TempDir() + bad := testItem(t, 7, "../../escape.conf", 10) + + seq := func(yield func(tgsource.Item, error) bool) { yield(bad, nil) } + it := newElemIter(seq, staging, false) + defer func() { _ = it.Close() }() + + if it.Next(t.Context()) { + t.Fatal("iterator accepted a name that escapes the staging directory") + } + err := it.Err() + if err == nil { + t.Fatal("Err() = nil after rejecting an unsafe name") + } + if !strings.Contains(err.Error(), "message 7") { + t.Errorf("error should name the message, got: %v", err) + } +} + +func TestElemIterOpensPartFilesAndPropagatesWalkErrors(t *testing.T) { + staging := t.TempDir() + good := testItem(t, 1, "a.mp4", 10) + + t.Run("opens a part file", func(t *testing.T) { + seq := func(yield func(tgsource.Item, error) bool) { yield(good, nil) } + it := newElemIter(seq, staging, true) + defer func() { _ = it.Close() }() + + if !it.Next(t.Context()) { + t.Fatalf("Next() = false, Err() = %v", it.Err()) + } + e := it.Value() + if !e.AsTakeout() { + t.Error("AsTakeout() = false, want the configured value") + } + if e.File().Size() != 10 || e.File().DC() != 2 { + t.Errorf("File() = size %d dc %d, want 10 and 2", e.File().Size(), e.File().DC()) + } + if _, err := os.Stat(partPath(staging, good)); err != nil { + t.Errorf("part file was not created: %v", err) + } + }) + + t.Run("propagates a walk error", func(t *testing.T) { + want := errors.New("history walk failed") + seq := func(yield func(tgsource.Item, error) bool) { yield(tgsource.Item{}, want) } + it := newElemIter(seq, staging, false) + defer func() { _ = it.Close() }() + + if it.Next(t.Context()) { + t.Fatal("Next() = true after a walk error") + } + if !errors.Is(it.Err(), want) { + t.Errorf("Err() = %v, want %v", it.Err(), want) + } + }) +} + +// Byte accounting has to treat ProgressState as a running total, not a delta, +// or the aggregate drifts upward on every callback. +func TestProgressAccountsBytesAsRunningTotals(t *testing.T) { + staging := t.TempDir() + it := testItem(t, 1, "a.mp4", 100) + e := openElem(t, staging, it) + + p := newProgress(func(*elem, error) error { return nil }, nil) + p.OnAdd(e) + p.OnDownload(e, progressState(40)) + p.OnDownload(e, progressState(100)) + p.OnDone(e, nil) + + outcomes, stats := p.results() + if stats.BytesDone != 100 { + t.Errorf("BytesDone = %d, want 100 (states are totals, not deltas)", stats.BytesDone) + } + if stats.BytesTotal != 100 { + t.Errorf("BytesTotal = %d, want 100", stats.BytesTotal) + } + if stats.Done != 1 || stats.Failed != 0 { + t.Errorf("Done/Failed = %d/%d, want 1/0", stats.Done, stats.Failed) + } + if len(outcomes) != 1 || outcomes[0].Err != nil { + t.Errorf("outcomes = %+v, want one success", outcomes) + } +} + +func TestProgressRecordsFailuresWithoutAborting(t *testing.T) { + staging := t.TempDir() + ok := testItem(t, 1, "a.mp4", 10) + bad := testItem(t, 2, "b.mp4", 10) + + p := newProgress(func(*elem, error) error { return nil }, nil) + p.OnDone(openElem(t, staging, ok), nil) + p.OnDone(openElem(t, staging, bad), errors.New("flood wait")) + + outcomes, stats := p.results() + if stats.Done != 1 || stats.Failed != 1 { + t.Errorf("Done/Failed = %d/%d, want 1/1", stats.Done, stats.Failed) + } + if len(outcomes) != 2 { + t.Fatalf("outcomes = %d, want 2 — a failure must be recorded, not dropped", len(outcomes)) + } +} + +func progressState(done int64) downloader.ProgressState { + return downloader.ProgressState{Downloaded: done, Total: 100} +} + +// core's Download logs a failed transfer and returns nil, and OnDone is deferred +// on that named return — so a truncated file reaches finish claiming success. +// The size check is the only thing standing between that and an archived +// fragment, so it is tested directly. +func TestFinishRejectsShortDownloadDespiteNilError(t *testing.T) { + staging := t.TempDir() + it := testItem(t, 3, "video.mp4", 1000) + + e := openElem(t, staging, it) + if _, err := e.file.WriteAt(make([]byte, 400), 0); err != nil { + t.Fatalf("write: %v", err) + } + + // nil, exactly as the downloader reports a failed transfer. + err := finish(staging, e, nil) + if err == nil { + t.Fatal("finish accepted a 400-byte file for a 1000-byte item") + } + if !strings.Contains(err.Error(), "short download") { + t.Errorf("error should name the short download, got: %v", err) + } + if _, err := os.Stat(finalPath(staging, it)); !os.IsNotExist(err) { + t.Error("a truncated file was promoted to its final name") + } + if _, err := os.Stat(partPath(staging, it)); !os.IsNotExist(err) { + t.Error("the truncated part file was left behind") + } +} + +// The size check must not reject a genuinely complete file. +func TestFinishAcceptsExactSize(t *testing.T) { + staging := t.TempDir() + it := testItem(t, 4, "exact.mp4", 2048) + + e := openElem(t, staging, it) + if _, err := e.file.WriteAt(make([]byte, 2048), 0); err != nil { + t.Fatalf("write: %v", err) + } + if err := finish(staging, e, nil); err != nil { + t.Fatalf("finish rejected an exact-size file: %v", err) + } + info, err := os.Stat(finalPath(staging, it)) + if err != nil { + t.Fatalf("final file missing: %v", err) + } + if info.Size() != 2048 { + t.Errorf("final size = %d, want 2048", info.Size()) + } +} diff --git a/internal/pipeline/elem.go b/internal/pipeline/elem.go index 30e39fa..d3c3464 100644 --- a/internal/pipeline/elem.go +++ b/internal/pipeline/elem.go @@ -64,6 +64,12 @@ type elemIter struct { staging string takeout bool + // acquire reserves staging space for the next item. Blocking here is what + // makes backpressure work: core's Download calls Next from its dispatch + // loop, so a blocked Next stops new downloads starting without stopping the + // uploads that free the space. + acquire func(context.Context, int64) error + current *elem err error @@ -104,6 +110,13 @@ func (i *elemIter) Next(ctx context.Context) bool { return false } + if i.acquire != nil { + if err := i.acquire(ctx, item.Size()); err != nil { + i.err = err + return false + } + } + f, err := os.OpenFile(partPath(i.staging, item), os.O_CREATE|os.O_RDWR, 0o600) if err != nil { i.err = fmt.Errorf("open destination for message %d: %w", item.MessageID, err) diff --git a/internal/pipeline/pipeline.go b/internal/pipeline/pipeline.go new file mode 100644 index 0000000..2bcd1b4 --- /dev/null +++ b/internal/pipeline/pipeline.go @@ -0,0 +1,193 @@ +package pipeline + +import ( + "context" + "errors" + "fmt" + "iter" + "sync" + + "github.com/rclone/rclone/fs" + "golang.org/x/sync/semaphore" + + "github.com/iyear/tdl/core/dcpool" + + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" +) + +// Options configures a full download-and-upload run. +type Options struct { + Pool dcpool.Pool + Dst fs.Fs + Staging string + + Threads int // connections per file + Limit int // files downloading at once + Uploads int // files uploading at once + Budget int64 // bytes allowed in staging at once; 0 means unbounded + + Confirm bool // re-state each uploaded object to prove its size + Takeout bool + + // MaxFailures trips the run after this many consecutive upload failures. + // Zero uses the shell pipeline's default of 5. + MaxFailures int + + Report func(Stats) +} + +// Result is what a run achieved. +type Result struct { + Stats Stats + Outcomes []Outcome +} + +// Failed lists the items that did not make it to the remote. +func (r Result) Failed() []Outcome { + var out []Outcome + for _, o := range r.Outcomes { + if o.Err != nil { + out = append(out, o) + } + } + return out +} + +// Run downloads every item and uploads each one as it completes. +// +// This is the whole reason for the rewrite. run.sh could not see inside tdl, so +// it inferred completion from a filename suffix plus a file's age, polled +// `du -sk` every ten seconds, and enforced its disk cap by sending SIGSTOP and +// SIGCONT to the tdl process. None of that exists here. Completion is a function +// returning. The cap is a semaphore: a download acquires its own size before +// starting and releases it only once the upload has confirmed, so when the +// remote is slow the acquire blocks and downloads pause on their own. +// +// Blocking in the iterator is safe by construction — core's Download calls +// Iter.Next from its dispatch loop while workers run in an errgroup, so a +// blocked Next stalls new work without stopping the uploads that free the budget +// (downloader.go:36-63). +func Run(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Options) (Result, error) { + if o.Uploads <= 0 { + o.Uploads = 1 + } + if o.MaxFailures <= 0 { + o.MaxFailures = 5 + } + + local, err := fs.NewFs(ctx, o.Staging) + if err != nil { + return Result{}, fmt.Errorf("open staging directory as a filesystem: %w", err) + } + up := &uploader{local: local, dst: o.Dst, confirm: o.Confirm} + + budget := newBudget(o.Budget) + uploads := make(chan tgsource.Item, o.Uploads) + + // Upload workers own the release side of the budget, so every path out of + // one — success, failure, cancellation — must release, or the run deadlocks + // with downloads waiting on space that is never freed. + var ( + wg sync.WaitGroup + mu sync.Mutex + uploadErrs []error + streak int + tripped bool + ) + upCtx, tripRun := context.WithCancel(ctx) + defer tripRun() + + for range o.Uploads { + wg.Add(1) + go func() { + defer wg.Done() + for it := range uploads { + err := up.upload(upCtx, it) + budget.release(it.Size()) + + mu.Lock() + if err != nil { + uploadErrs = append(uploadErrs, err) + streak++ + if streak >= o.MaxFailures && !tripped { + // A remote that fails this many times running is not + // going to recover on its own, and continuing just fills + // staging until the disk does. + tripped = true + tripRun() + } + } else { + streak = 0 + } + mu.Unlock() + } + }() + } + + dlOutcomes, stats, dlErr := Download(ctx, seq, DownloadOptions{ + Pool: o.Pool, + Staging: o.Staging, + Threads: o.Threads, + Limit: o.Limit, + Takeout: o.Takeout, + Report: o.Report, + acquire: budget.acquire, + onReady: func(it tgsource.Item) { uploads <- it }, + onFailed: func(it tgsource.Item) { + // Nothing was staged, so the reservation has to come back here + // instead of from an upload that will never happen. + budget.release(it.Size()) + }, + }) + + close(uploads) + wg.Wait() + + mu.Lock() + errs := append([]error(nil), uploadErrs...) + trip := tripped + mu.Unlock() + + res := Result{Stats: stats, Outcomes: dlOutcomes} + switch { + case dlErr != nil: + return res, dlErr + case trip: + return res, fmt.Errorf("stopping after %d consecutive upload failures: %w", + o.MaxFailures, errors.Join(errs...)) + case len(errs) > 0: + return res, errors.Join(errs...) + } + return res, nil +} + +// budget bounds how many bytes of downloaded-but-not-yet-uploaded data sit on +// local disk. A zero limit means no bound. +type budget struct{ sem *semaphore.Weighted } + +func newBudget(limit int64) *budget { + if limit <= 0 { + return &budget{} + } + return &budget{sem: semaphore.NewWeighted(limit)} +} + +func (b *budget) acquire(ctx context.Context, n int64) error { + if b.sem == nil { + return nil + } + // An item larger than the whole budget could never be admitted and would + // block forever, so it is refused with an error that says what to change. + // Callers validate up front too; this is the guard for an item whose size + // was not known then. + if err := b.sem.Acquire(ctx, n); err != nil { + return fmt.Errorf("waiting for %d bytes of staging space: %w", n, err) + } + return nil +} + +func (b *budget) release(n int64) { + if b.sem != nil { + b.sem.Release(n) + } +} diff --git a/internal/pipeline/pipeline_test.go b/internal/pipeline/pipeline_test.go new file mode 100644 index 0000000..e36271e --- /dev/null +++ b/internal/pipeline/pipeline_test.go @@ -0,0 +1,147 @@ +package pipeline + +import ( + "context" + "errors" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/iyear/tdl/core/tmedia" + + "github.com/tiennm99dev/telegram-exporter/internal/naming" + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" +) + +// The budget is what replaces run.sh's du-polling and SIGSTOP/SIGCONT cap +// draining, so the property it has to hold is simple and worth pinning: the sum +// of outstanding reservations never exceeds the limit. +func TestBudgetBoundsOutstandingBytes(t *testing.T) { + const limit = 1000 + b := newBudget(limit) + ctx := t.Context() + + var ( + mu sync.Mutex + held int64 + peak int64 + wg sync.WaitGroup + acquires atomic.Int64 + ) + + for range 20 { + wg.Add(1) + go func() { + defer wg.Done() + const size = 300 + if err := b.acquire(ctx, size); err != nil { + t.Errorf("acquire: %v", err) + return + } + acquires.Add(1) + + mu.Lock() + held += size + if held > peak { + peak = held + } + mu.Unlock() + + time.Sleep(time.Millisecond) + + mu.Lock() + held -= size + mu.Unlock() + b.release(size) + }() + } + wg.Wait() + + if acquires.Load() != 20 { + t.Errorf("acquired %d times, want 20 — every item must eventually get through", acquires.Load()) + } + if peak > limit { + t.Errorf("peak outstanding = %d bytes, over the %d limit", peak, limit) + } +} + +// A zero limit means the operator asked for no cap; acquiring must not block or +// account, or an unbounded run would stall. +func TestBudgetUnboundedWhenLimitIsZero(t *testing.T) { + b := newBudget(0) + for range 5 { + if err := b.acquire(t.Context(), 1<<40); err != nil { + t.Fatalf("acquire on an unbounded budget: %v", err) + } + } + b.release(1 << 40) // must not panic +} + +// Cancelling must unblock a waiter rather than leaving the run wedged. +func TestBudgetAcquireHonoursCancellation(t *testing.T) { + b := newBudget(100) + if err := b.acquire(t.Context(), 100); err != nil { + t.Fatalf("first acquire: %v", err) + } + + ctx, cancel := context.WithCancel(t.Context()) + done := make(chan error, 1) + go func() { done <- b.acquire(ctx, 100) }() + + // The second acquire cannot succeed while the first is outstanding. + select { + case err := <-done: + t.Fatalf("acquire succeeded with no space free: %v", err) + case <-time.After(50 * time.Millisecond): + } + + cancel() + select { + case err := <-done: + if !errors.Is(err, context.Canceled) { + t.Errorf("acquire error = %v, want context.Canceled", err) + } + case <-time.After(2 * time.Second): + t.Fatal("cancelling did not unblock the waiter") + } +} + +// An item bigger than the whole budget can never be admitted. It must fail +// rather than hang, because a hang here looks exactly like a slow remote. +func TestBudgetRefusesAnItemLargerThanTheLimit(t *testing.T) { + b := newBudget(100) + ctx, cancel := context.WithTimeout(t.Context(), 500*time.Millisecond) + defer cancel() + + err := b.acquire(ctx, 5000) + if err == nil { + t.Fatal("acquire of an oversized item succeeded") + } + if !errors.Is(err, context.DeadlineExceeded) { + t.Errorf("acquire error = %v, want the deadline to surface", err) + } +} + +func TestResultFailedListsOnlyErrors(t *testing.T) { + r := Result{Outcomes: []Outcome{ + {Item: mustItem(1), Err: nil}, + {Item: mustItem(2), Err: errors.New("flood wait")}, + {Item: mustItem(3), Err: nil}, + {Item: mustItem(4), Err: errors.New("short download")}, + }} + failed := r.Failed() + if len(failed) != 2 { + t.Fatalf("Failed() = %d entries, want 2", len(failed)) + } + for _, f := range failed { + if f.Err == nil { + t.Errorf("Failed() returned a successful outcome: %+v", f) + } + } +} + +func mustItem(id int) tgsource.Item { + m := &tmedia.Media{Name: "f.mp4", Size: 10} + return tgsource.Item{DialogID: 1, MessageID: id, Name: naming.For(1, id, m), Media: m} +} diff --git a/internal/pipeline/progress.go b/internal/pipeline/progress.go new file mode 100644 index 0000000..8c7c75c --- /dev/null +++ b/internal/pipeline/progress.go @@ -0,0 +1,105 @@ +package pipeline + +import ( + "sync" + + "github.com/iyear/tdl/core/downloader" + + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" +) + +// Outcome is what happened to one item. +type Outcome struct { + Item tgsource.Item + Err error // nil when the file downloaded and was renamed into place +} + +// Stats is a snapshot of a run's progress. +type Stats struct { + Started int + Done int + Failed int + BytesDone int64 + BytesTotal int64 +} + +// progress collects per-item results and feeds an optional live reporter. +// +// The downloader calls these from its worker goroutines, so everything here is +// mutex-guarded. OnDone fires once per item whether it succeeded or not, which +// is what makes it the right place to finish the file — the downloader itself +// never closes or renames what To() handed it. +type progress struct { + mu sync.Mutex + stats Stats + outcomes []Outcome + inFlight map[int]int64 // message id -> bytes written so far + + finish func(*elem, error) error + report func(Stats) +} + +func newProgress(finish func(*elem, error) error, report func(Stats)) *progress { + return &progress{ + inFlight: make(map[int]int64), + finish: finish, + report: report, + } +} + +func (p *progress) OnAdd(e downloader.Elem) { + el := e.(*elem) + p.mu.Lock() + p.stats.Started++ + p.stats.BytesTotal += el.item.Size() + stats := p.stats + p.mu.Unlock() + p.emit(stats) +} + +func (p *progress) OnDownload(e downloader.Elem, state downloader.ProgressState) { + el := e.(*elem) + p.mu.Lock() + // State carries the running total for this item, not a delta, so the + // aggregate is adjusted by the difference since the last callback. + prev := p.inFlight[el.item.MessageID] + p.inFlight[el.item.MessageID] = state.Downloaded + p.stats.BytesDone += state.Downloaded - prev + stats := p.stats + p.mu.Unlock() + p.emit(stats) +} + +func (p *progress) OnDone(e downloader.Elem, err error) { + el := e.(*elem) + + // Closing and renaming happens here because this is the only callback that + // runs exactly once per item and knows whether it succeeded. + if ferr := p.finish(el, err); ferr != nil && err == nil { + err = ferr + } + + p.mu.Lock() + delete(p.inFlight, el.item.MessageID) + if err != nil { + p.stats.Failed++ + } else { + p.stats.Done++ + } + p.outcomes = append(p.outcomes, Outcome{Item: el.item, Err: err}) + stats := p.stats + p.mu.Unlock() + p.emit(stats) +} + +func (p *progress) emit(s Stats) { + if p.report != nil { + p.report(s) + } +} + +func (p *progress) results() ([]Outcome, Stats) { + p.mu.Lock() + defer p.mu.Unlock() + return p.outcomes, p.stats +} diff --git a/internal/pipeline/upload.go b/internal/pipeline/upload.go new file mode 100644 index 0000000..89ed3eb --- /dev/null +++ b/internal/pipeline/upload.go @@ -0,0 +1,47 @@ +package pipeline + +import ( + "context" + "fmt" + + "github.com/rclone/rclone/fs" + "github.com/rclone/rclone/fs/operations" + + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" +) + +// uploader moves finished files from staging to the destination remote. +type uploader struct { + local fs.Fs // the staging directory as an rclone filesystem + dst fs.Fs + confirm bool +} + +// upload moves one finished file to the remote and, unless disabled, proves it +// arrived at the expected size. +// +// MoveFile removes the local copy as part of the move, so a successful return +// means the file is on the remote and off local disk — which is what lets the +// byte budget be released. +// +// Confirmation closes a gap the shell pipeline left open: there, a truncated +// upload was only noticed by a later verify pass, after the local copy was +// already gone. Re-stating the object costs one round trip per file and turns a +// silent corruption into a retry. +func (u *uploader) upload(ctx context.Context, it tgsource.Item) error { + if err := operations.MoveFile(ctx, u.dst, u.local, it.Name, it.Name); err != nil { + return fmt.Errorf("move %q to %s: %w", it.Name, u.dst.String(), err) + } + if !u.confirm { + return nil + } + + obj, err := u.dst.NewObject(ctx, it.Name) + if err != nil { + return fmt.Errorf("confirm %q: %w", it.Name, err) + } + if got := obj.Size(); got != it.Size() { + return fmt.Errorf("confirm %q: remote has %d bytes, expected %d", it.Name, got, it.Size()) + } + return nil +} diff --git a/internal/remote/fs.go b/internal/remote/fs.go index 6e71596..1ab57b0 100644 --- a/internal/remote/fs.go +++ b/internal/remote/fs.go @@ -98,6 +98,20 @@ func Resolve(ctx context.Context, remote string) (fs.Fs, error) { return f, nil } +// EnsureDir creates the destination if it is not there yet. +// +// A destination that does not exist yet is the normal case for a first run, and +// the shell pipeline created it up front for the same reason — the call doubles +// as the reachability and credentials check, since a remote that refuses a +// mkdir will refuse the uploads too. Doing it before the chat is read means a +// bad destination fails in seconds rather than after a full history walk. +func EnsureDir(ctx context.Context, f fs.Fs) error { + if err := f.Mkdir(ctx, ""); err != nil { + return fmt.Errorf("cannot create %q — check credentials and connectivity: %w", f.String(), err) + } + return nil +} + // FreeBytes reports free space on the remote. // // Backends without quota reporting return ok=false rather than an error: the diff --git a/internal/remote/index.go b/internal/remote/index.go index 73bd950..40acc9c 100644 --- a/internal/remote/index.go +++ b/internal/remote/index.go @@ -2,6 +2,7 @@ package remote import ( "context" + "errors" "fmt" "path" "slices" @@ -88,6 +89,12 @@ func BuildIndex(ctx context.Context, f fs.Fs, dialogID int64) (*Index, error) { idx.byID[id] = append(idx.byID[id], name) } }); err != nil { + // A destination that does not exist yet holds nothing. That is an empty + // index, not a failure — it is what a first run against a new path looks + // like, and treating it as an error would make verify unusable there. + if errors.Is(err, fs.ErrorDirNotFound) { + return idx, nil + } return nil, fmt.Errorf("list %s: %w", f.String(), err) } return idx, nil diff --git a/internal/report/progress.go b/internal/report/progress.go new file mode 100644 index 0000000..e458fa8 --- /dev/null +++ b/internal/report/progress.go @@ -0,0 +1,108 @@ +// Package report renders a run's progress for whoever is watching. +package report + +import ( + "fmt" + "io" + "os" + "sync" + "time" + + "github.com/tiennm99dev/telegram-exporter/internal/pipeline" +) + +// statsInterval is how often a redirected run prints a line. +// +// The shell pipeline learned this the hard way: tdl's progress bar is ANSI +// redraws, which are right on a terminal and turn a captured log into +// megabytes of control characters. So a TTY gets a redrawn line and everything +// else gets a periodic summary. +const statsInterval = 30 * time.Second + +// Reporter renders progress, adapting to whether it is writing to a terminal. +type Reporter struct { + w io.Writer + tty bool + total int + totalBytes int64 + + mu sync.Mutex + started time.Time + lastLine time.Time +} + +// New builds a reporter for w, which is treated as a terminal when it is one. +// +// The totals are passed in rather than taken from Stats because Stats.BytesTotal +// only counts items the downloader has started, so a progress line built from it +// shows a denominator that grows as the run proceeds — "0 B of 52 KiB" on a run +// that will move gigabytes. The caller knows the real figures before starting. +func New(w io.Writer, total int, totalBytes int64) *Reporter { + return &Reporter{w: w, tty: isTerminal(w), total: total, totalBytes: totalBytes, started: time.Now()} +} + +// Update renders a snapshot. Safe to call from several goroutines, and cheap +// enough to call on every progress callback. +func (r *Reporter) Update(s pipeline.Stats) { + r.mu.Lock() + defer r.mu.Unlock() + + now := time.Now() + if r.tty { + // \r rather than \n: one line, redrawn. + fmt.Fprintf(r.w, "\r\033[K%s", r.line(s, now)) + return + } + if now.Sub(r.lastLine) < statsInterval { + return + } + r.lastLine = now + fmt.Fprintf(r.w, "%s\n", r.line(s, now)) +} + +// Finish writes the closing summary, ending the redrawn line if there was one. +func (r *Reporter) Finish(s pipeline.Stats) { + r.mu.Lock() + defer r.mu.Unlock() + if r.tty { + fmt.Fprint(r.w, "\r\033[K") + } + elapsed := time.Since(r.started).Round(time.Second) + fmt.Fprintf(r.w, "%d done, %d failed, %s in %s (%s/s)\n", + s.Done, s.Failed, humanBytes(s.BytesDone), elapsed, + humanBytes(int64(float64(s.BytesDone)/max(elapsed.Seconds(), 1)))) +} + +func (r *Reporter) line(s pipeline.Stats, now time.Time) string { + elapsed := now.Sub(r.started) + rate := float64(s.BytesDone) / max(elapsed.Seconds(), 1) + return fmt.Sprintf("%d/%d done, %d failed, %s of %s, %s/s", + s.Done, r.total, s.Failed, + humanBytes(s.BytesDone), humanBytes(r.totalBytes), humanBytes(int64(rate))) +} + +func humanBytes(n int64) string { + const unit = 1024 + if n < unit { + return fmt.Sprintf("%d B", n) + } + div, exp := int64(unit), 0 + for v := n / unit; v >= unit; v /= unit { + div *= unit + exp++ + } + return fmt.Sprintf("%.1f %ciB", float64(n)/float64(div), "KMGT"[exp]) +} + +// isTerminal reports whether w is a character device. +func isTerminal(w io.Writer) bool { + f, ok := w.(*os.File) + if !ok { + return false + } + info, err := f.Stat() + if err != nil { + return false + } + return info.Mode()&os.ModeCharDevice != 0 +} From bd816156ebc216a8b6ffa76d6c4a5c0bd8497379 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Sun, 6 Sep 2026 19:37:30 +0700 Subject: [PATCH 05/16] fix: join download workers before closing the upload channel core's Download returns without waiting on its worker group when the iterator reports an error, so surfacing one through Iter.Err left workers sending into a channel the caller had already closed. The iterator now always reports a nil error and stashes the real one, read after Download returns. The circuit breaker cancelled only the upload context, which left downloads running full speed against a remote refusing them: every file stayed in staging and every reservation came back, so a broken remote filled local disk faster than a working one. It now stops the download iterator instead. A failed move leaves the local copy in place, so the byte reservation cannot be handed back until the file is removed. A confirmed-short object is deleted rather than left under a name verification would count as archived forever. Also: release the reservation when opening the staging file fails, report results even when an upload errored, bound the recorded errors, and drop the reporter's lock before writing so terminal latency cannot throttle downloads. --- README.md | 399 ++++++++++------------------- cmd/tgexport/sync.go | 38 +-- go.mod | 2 +- internal/pipeline/download.go | 32 ++- internal/pipeline/download_test.go | 123 ++++++++- internal/pipeline/elem.go | 69 +++-- internal/pipeline/pipeline.go | 94 ++++--- internal/pipeline/upload.go | 27 +- internal/report/progress.go | 13 +- 9 files changed, 439 insertions(+), 358 deletions(-) diff --git a/README.md b/README.md index b22174f..5cc42ae 100644 --- a/README.md +++ b/README.md @@ -1,305 +1,176 @@ # telegram-exporter -Export a Telegram chat's media to **any rclone remote** — S3, Google Drive, -Dropbox, Backblaze B2, SFTP, WebDAV, or anything else rclone supports — using -far less local disk than the chat's total size. +Archive a Telegram chat's media to **any rclone remote** — S3, Google Drive, +Dropbox, Backblaze B2, SFTP, WebDAV, pikpak, or anything else rclone supports — +using far less local disk than the chat's total size. -`run.sh` runs [tdl](https://github.com/iyear/tdl) and -[rclone](https://rclone.org/) as a rolling pipeline: tdl downloads into a small -staging directory while rclone concurrently moves finished files to the remote -and deletes the local copies. Local disk only ever holds the files in flight -plus one sync interval of throughput, so a multi-terabyte chat exports fine on a -small disk. Telegram caps a single file at 2 GB (4 GB from premium uploaders), -so a few dozen GB of staging covers the worst case regardless of chat size. +`tgexport` embeds [tdl](https://github.com/iyear/tdl) and +[rclone](https://rclone.org/) as libraries and runs both halves in one process. +Files are downloaded into a small staging directory and uploaded the moment each +one finishes, so local disk only ever holds what is in flight. A multi-terabyte +chat archives fine on a small disk. Telegram caps a single file at 2 GB (4 GB +from premium uploaders), so a few dozen GB of staging covers the worst case +regardless of chat size. -This is needed because **tdl can only write to a local directory** — it has no -rclone integration and no remote destination of any kind (`tdl dl -d` takes a -filesystem path; `tdl --storage` is its session database, not an output target). -tdl downloads over MTProto with a user account, so Bot API limits do not apply: -full history is readable and there is no 20 MB download cap. +This exists because **tdl can only write to a local directory** — it has no +remote destination of any kind (`tdl dl -d` takes a filesystem path; `tdl +--storage` is its session database, not an output target). tdl downloads over +MTProto with a user account, so Bot API limits do not apply: full history is +readable and there is no 20 MB download cap. ## Requirements -- **tdl** — -- **rclone** — -- **bash**. On Windows, run under WSL or Git Bash. +- **Go 1.25+** to build, or a prebuilt binary. +- **tdl** — only for `tdl login`. +- **rclone** — only to configure a remote. + +Neither tool is invoked at run time; `tgexport` reads the session and config +they write. ## Setup Both steps are one-time. ```bash -# 1) log in to Telegram with your user account (phone + code + 2FA) -tdl login - -# 2) configure the destination. The interactive wizard covers every backend: -rclone config - -# ...or create one non-interactively, e.g. -rclone config create gdrive drive -rclone config create b2 b2 account=KEY_ID key=APP_KEY -rclone config create dav webdav url=https://dav.example.com/remote.php/dav/files/you \ - vendor=other user=YOU pass=SECRET - -rclone listremotes # confirm the name you will pass to -r +tdl login # writes the Telegram session tgexport reads +rclone config # define the destination remote +go build -o tgexport ./cmd/tgexport +./tgexport doctor -r myremote:archive ``` -Any rclone remote form works, including on-the-fly connection strings -(`:webdav,url=https://...:/path`). +`doctor` proves both halves work before a long run: it prints the logged-in +account, resolves the destination, and reports free space. + +### Build variants + +| Build | Backends | Size | +|---|---|---| +| `go build ./cmd/tgexport` | every rclone backend | ~92 MB | +| `go build -tags slim ./cmd/tgexport` | pikpak only | ~49 MB | + +A backend that is not compiled in does not exist at run time, so use the default +build unless the destination will never change. ## Usage ```bash -./run.sh -r gdrive:telegram/media -c @mygroup +./tgexport sync -c CHAT -r REMOTE:PATH [options] ``` -That is the whole flow. It exports the chat's message metadata to -`export.json`, then downloads and uploads concurrently until finished. Progress -and warnings go to stderr; press Ctrl-C at any point and it stops cleanly. +`CHAT` accepts a numeric id as printed by `tdl chat ls`, a username with or +without `@`, or a `t.me`/`tg://` link. A Bot API `-100…` id is converted +automatically. A link to a single *message* is refused — it names a message, not +a chat. + +```bash +# archive a chat, capping staging at 40 GiB +./tgexport sync -c @mychannel -r gdrive:telegram/media -m 40G + +# check completeness without downloading anything +./tgexport verify -c @mychannel -r gdrive:telegram/media + +# list what the chat holds +./tgexport list -c @mychannel +``` ### Options -| Flag | Meaning | -|------|---------| -| `-r REMOTE:PATH` | **Required.** rclone destination, e.g. `gdrive:telegram/media`, `s3:bucket/tg`, `dav:tg-export` | -| `-c CHAT` | Chat to export when the JSON does not exist yet — id, username, or link (see below) | -| `-f FILE` | Export JSON to download from (default `export-.json` with `-c`, else `export.json`) | -| `-d DIR` | Staging directory (default `./staging`) | -| `-i SECONDS` | Seconds between rclone sweeps (default `60`) | -| `-a AGE` | rclone `--min-age`, a second guard against moving files still being written (default `45s`) | -| `-m SIZE` | Cap the staging directory at `SIZE` (`K`/`M`/`G`/`T`, binary), e.g. `40G`. Unset means no cap (see below) | -| `-h` | Help | +| Flag | Default | Meaning | +|---|---|---| +| `-c` | — | chat id, username, or link (required) | +| `-r` | — | rclone destination, `REMOTE:PATH` (required) | +| `-d` | `./staging` | staging directory for files in flight | +| `-m` | no cap | cap staging at a size, e.g. `40G` | +| `--threads` | 4 | connections per file | +| `--limit` | 2 | files downloading at once | +| `--uploads` | 2 | files uploading at once | +| `--min-free` | 5 | stop if the remote has fewer than this many GiB free | +| `--limit-items` | 0 | stop after N files; for smoke tests | +| `--confirm` | true | re-state each uploaded file to prove its size | +| `--takeout` | true | use a takeout session | +| `-n` | `default` | tdl session namespace | -### Identifying the chat +### Exit codes -`-c` accepts every form tdl understands, plus one it doesn't: +| Code | Meaning | +|---|---| +| 0 | complete | +| 1 | ran, but files remain | +| 2 | usage error | +| 3 | remote or Telegram failure | +| 130 / 143 | interrupted (SIGINT / SIGTERM) | -| Form | Example | -|------|---------| -| Numeric id, as printed by `tdl chat ls` | `-c 1697797156` | -| Username, with or without `@` | `-c @mygroup` / `-c mygroup` | -| Public link | `-c https://t.me/mygroup` / `-c t.me/mygroup` | -| Deep link | `-c 'tg://resolve?domain=mygroup'` | -| **Bot API id** (converted for you) | `-c -1001697797156` → `1697797156` | +## How it works -tdl resolves a numeric argument as an MTProto id and anything else through -gotd's resolver. MTProto has no `-100` prefix, so a Bot API id would otherwise -fail to resolve; the script strips it and logs the conversion. +Re-running is the resume path. Each item is checked against a listing of the +remote immediately before download, so an interrupted run picks up where it left +off and a completed one downloads nothing. -A **message** link is rejected — `-c` names a chat, not a message: +**Filenames.** Every file is stored as `{DialogID}_{MessageID}_{FileName}`, where +`FileName` is exactly what Telegram reports. One function derives that string, +and the same string is used both to ask whether the file is already archived and +to write it — so the two can never disagree. -``` -$ ./run.sh -r gdrive:tg -c https://t.me/mygroup/123 -error: -c takes a chat, not a message link — pass the chat's username or id -``` +That last point is the reason this program exists. Its predecessor derived the +name twice: `tdl chat export` wrote the raw name into a JSON, while `tdl dl` +rendered it through a template applying `filenamify`, which rewrites characters a +filesystem rejects and collapses runs of `!`. A file whose name contained `!!` +was looked up under one name and stored under another, so the verifier never +found it and re-fetched it on every pass — forever, at 966 MB a time. -Run `tdl chat ls` to see ids and usernames side by side. +Note the consequence: names are **not** run through `filenamify`, so they are not +byte-compatible with what the old shell pipeline wrote. A file it stored under a +rewritten name will not be recognised and gets fetched again. -Each chat gets its own export file by default (`export-mygroup.json`, -`export-1697797156.json`), so exporting a second chat from the same directory -never reuses the first one's JSON. When the file already exists it is reused and -the script says so — delete it to re-export. +**Disk.** `-m` is a byte budget. A download reserves its own size before starting +and releases it only once the upload is confirmed, so when the remote is slow the +downloads pause on their own. The cap must exceed the largest single file, and a +cap that does not is refused at startup rather than discovered as a hang. -Anything after `--` is passed straight to `tdl dl`: +**Integrity.** A download is written to `.part` and renamed only once its +size matches what Telegram reported, so a file without the suffix is always +whole. Uploads are re-stated afterwards to prove they arrived at the right size, +before the local copy is gone. -```bash -./run.sh -r gdrive:telegram/media -- -t 4 -l 1 # calmer parallelism, fewer flood waits -./run.sh -r gdrive:telegram/media -- -i mp4,mkv # only these file extensions -./run.sh -r gdrive:telegram/media -- -e jpg,png # skip these file extensions -``` +## Replacing the shell pipeline -tdl defaults to `-t 8 -l 4`, which is aggressive; lower it if you hit flood -waits on a large export. +Earlier versions of this repo were three bash scripts — `run.sh`, +`export-until-complete.sh` and `verify-export.sh` — driving `tdl` and `rclone` as +separate processes. Everything expensive in them existed to work around the fact +that neither process could see the other's state: a staging directory polled with +`du -sk`, an `--min-age` guard, a `*.tmp` exclusion, `SIGSTOP`/`SIGCONT` to +enforce the disk cap, a sweep-failure counter, and an outer loop that re-verified +and re-narrowed a JSON export between passes. -### Tuning the upload +One process needs none of it. Completion is a function returning; the cap is a +semaphore. Some hard-won details were worth keeping, and are: -rclone reads every one of its flags from an environment variable, so the upload -side is tunable without touching the script: +- **pikpak commits uploads as a server-side async task**, and rclone abandons a + still-pending one when its low-level retries run out. `transfers=2` and + `low-level-retries=20` are the defaults here for that reason. Environment + overrides still win. +- **A backend with no quota API is treated as unlimited**, so it never blocks a + run. +- **Zero-byte files count as missing** — rclone overwrites a size-mismatched + destination, so re-running repairs them — while files under 1 KiB are reported + but trusted, since some real media genuinely is that small. -```bash -RCLONE_TRANSFERS=8 RCLONE_BWLIMIT=20M ./run.sh -r s3:bucket/tg -c @mygroup -``` +Flags that disappeared are recognised and explain what replaced them: -`run.sh` sets two of those itself, and only when the caller has not: +| Old | Why it is gone | +|---|---| +| `-i` | no sweep interval; uploads start when a download finishes | +| `-a` | no `--min-age`; completion is observed, not inferred | +| `-f` | no export JSON; the chat is read live, so names cannot go stale | +| `-p` | no passes; one invocation converges | +| `-q` | renamed `--min-free` | -| Variable | Default here | rclone's own default | Why | -|----------|--------------|----------------------|-----| -| `RCLONE_TRANSFERS` | `2` | `4` | Backends that commit an upload as a server-side async task queue those tasks; less parallelism keeps the queue short | -| `RCLONE_LOW_LEVEL_RETRIES` | `20` | `10` | Each retry re-polls a pending task, so a slow commit is waited out instead of failing the transfer | +## Notes -Both exist because of one failure mode. On pikpak an upload finishes in two -phases — rclone sends the bytes, then a server-side task must reach -`PHASE_TYPE_COMPLETE`. rclone waits 500 ms and then polls, giving up after -`--low-level-retries` attempts with: - -``` -ERROR : : Failed to copy: can't verify the task is completed: ... Phase:"PHASE_TYPE_PENDING" -``` - -Nothing is lost when that happens — the message is followed by `Not deleting -source as copy failed`, the file stays in staging and the next sweep retries -it. But it wastes the upload, and it counts against a `-m` cap, since a file -that keeps failing can never be drained. Raise the retries further if you still -see it. - -### Capping the staging directory - -Without `-m`, staging grows whenever tdl downloads faster than rclone uploads, -which on a fast connection and a slow remote can mean tens of GB between -sweeps. `-m` puts a ceiling on it: - -```bash -./run.sh -r s3:bucket/tg -c @mygroup -m 40G -``` - -Staging size is checked every 10 seconds, independently of `-i`. When it -reaches the cap, tdl is suspended with `SIGSTOP` and rclone sweeps until -staging is back under it, then tdl is resumed — it reconnects on its own and -`--continue` picks its `.tmp` files back up. Because the checks are periodic, -the cap is a high-water mark rather than a hard limit: staging can overshoot by -up to ten seconds of download throughput before the gate closes. - -Only finished files can be drained, so the cap has to exceed what the -concurrent downloads hold — at most `-l` times 2 GB (4 GB from premium -uploaders). With the default `-l 2` anything from ~10 GB up is safe; below -that, the drain cannot clear the cap and the run logs a warning on every check -instead of throttling. - -### Exporting a subset - -Generate the JSON yourself when you want a narrower export, then point `-f` at -it: - -```bash -tdl chat export -c @mygroup -T id -i 1000,5000 --all --with-content -o part.json -./run.sh -r gdrive:telegram/media -f part.json -``` - -`tdl chat export` takes `-T time|id|last` with `-i` as the range, and `-f` as an -expression filter over message fields (`-f -` lists the available fields). - -## Sweep output - -The periodic sweeps are silent — they run every `-i` seconds alongside tdl's own -output, and narrating each one would drown it. The sweeps that run **once at the -end** do report progress, since they can move the whole staging directory with -nothing else on screen: - -- the exit sweep on Ctrl-C, `SIGTERM`, or a tdl failure (`sweeping completed - files before exit`); -- the final sweep after tdl finishes successfully. - -On a terminal that is rclone's redrawn `--progress` bar. When output is -redirected to a log it becomes a one-line stats summary every 30s -(`--stats 30s --stats-one-line --stats-log-level NOTICE`) — rclone logs stats at -INFO, so raising just the stats to NOTICE avoids the line-per-file spam that -`-v` would add. - -To show progress on every sweep instead, rclone reads its flags from the -environment: - -```bash -RCLONE_PROGRESS=true ./run.sh -r s3:bucket/tg -c @mygroup -``` - -## Resuming - -Re-run the same command. Both legs resume independently and nothing is -downloaded or uploaded twice. - -Keep the same `export.json` between runs: `--skip-same` compares against the -**staging** directory, which is empty once files have moved to the remote, so -cross-run deduplication rests on tdl's own `--continue` tracking. If you must -start from a fresh export, narrow it to the missing message-id range -(`-T id -i ,`) rather than re-downloading everything. - -## Verifying an export - -`run.sh` finishes when tdl finishes, which is not the same as every file having -arrived: a dropped session, a stalled remote, or an interrupted pass all leave -gaps. `verify-export.sh` settles it by rebuilding the filename tdl produces for -each media message in the export JSON and checking the remote for it. - -```bash -./verify-export.sh -f export-mygroup.json -r remote:telegram/media -``` - -``` -messages in export : 18193 - text-only (skip) : 38 - media expected : 12000 -present and intact : 12000 - absent : 0 - zero-byte : 0 - -COMPLETE: every media message is present and non-empty. -``` - -Messages with no media are skipped; they carry an empty `file` and were never -download targets. A zero-byte file counts as missing, because rclone overwrites a -size-mismatched destination and a retry repairs it. Files under 1 KiB are -reported but not retried, since some real media is genuinely that small. Exit -status is 0 when complete and 1 otherwise, with the outstanding message ids -written to `missing-ids.txt`. - -## Running until complete - -`export-until-complete.sh` drives `run.sh` in a loop: verify what is already -there, narrow the export to the ids still missing, run the pipeline on that -subset, and repeat. - -```bash -./export-until-complete.sh -r remote:telegram/media -c @mygroup -``` - -| Flag | Meaning | -|------|---------| -| `-r REMOTE:PATH` | **Required.** rclone destination | -| `-c CHAT` | Chat to export metadata for on the first pass | -| `-f FILE` | Export JSON (default `export-.json`) | -| `-d DIR` | Staging directory (default `./staging`) | -| `-i SECONDS` | rclone sweep interval (default `120`) | -| `-m SIZE` | Staging cap passed through to `run.sh`, e.g. `40G` | -| `-p N` | Maximum passes (default `30`) | -| `-q GIB` | Stop if remote free space falls below this (default `5`) | - -It stops when the verifier reports complete (exit `0`), when a pass fetches -nothing new (exit `1` — the remaining media is no longer available from -Telegram), when the remote runs low on space (exit `3`), or on Ctrl-C (exit -`130`, after the current pass shuts down cleanly). - -tdl's progress bar is shown when stdout is a terminal and suppressed when output -is redirected, so a log file stays readable without a flag. - -## What it guards against - -- **Partial uploads.** tdl writes `.tmp` and renames on completion, so - every sweep excludes `*.tmp`. Age alone is not a completion signal: a download - stalled by a flood wait stops touching its `.tmp`, which would then be - uploaded half-written and lose its resume point. -- **Directories vanishing under tdl.** `--delete-empty-src-dirs` runs only in - the final sweep, and the staging directory is recreated after every sweep. - Removing a directory under a running tdl makes it fail to create its next file. -- **Orphaned downloads.** tdl is stopped on exit, Ctrl-C, or `SIGTERM`, so no - download keeps running after the script is gone. -- **A failed run looking finished.** The unrestricted final sweep happens only - after tdl exits 0. An interrupted or crashed run gets the age-guarded sweep - and keeps staging for the next attempt. -- **A dead remote filling the disk.** Five consecutive rclone failures abort the - run instead of letting staging grow unbounded. -- **A fast connection filling the disk.** With `-m`, tdl is suspended whenever - staging reaches the cap and resumed once rclone has drained it, so download - throughput cannot outrun the upload leg. -- **Typos and bad credentials.** Before downloading anything, the remote must be - present in `rclone listremotes` (skipped for connection strings) and the - destination must be creatable, which proves both reachability and auth. - -Exit codes: `0` success, `2` usage error, `3` rclone failure, `130`/`143` -interrupted, anything else is tdl's own exit code. - -## Limits - -- Streaming with no staging at all (piping download chunks straight to the - remote) is not possible with tdl and would require custom code. -- The script is bash; the two tools it drives are cross-platform, but Windows - needs WSL or Git Bash. +- `tgexport` and the `tdl` CLI share one session store and cannot run against the + same namespace at once. Use `-n` for a second namespace if you need both. +- A partially downloaded file is not resumable across restarts — tdl's library + exposes no resume offset — so an interrupted run re-fetches whatever was in + flight, bounded by `--limit`. +- Everything is read-only against Telegram. Nothing is uploaded, deleted, or + marked read. diff --git a/cmd/tgexport/sync.go b/cmd/tgexport/sync.go index 207b139..d28b726 100644 --- a/cmd/tgexport/sync.go +++ b/cmd/tgexport/sync.go @@ -36,27 +36,27 @@ var retiredFlags = map[string]string{ // syncCmd archives a chat to a remote: read the chat, skip what is already // there, download and upload the rest, then report on the result. func syncCmd(ctx context.Context, args []string) error { - fs := flag.NewFlagSet("sync", flag.ContinueOnError) + flags := flag.NewFlagSet("sync", flag.ContinueOnError) var ( - chat = fs.String("c", "", "chat id, username, or t.me link (required)") - remoteArg = fs.String("r", "", "rclone destination, e.g. pikpak:archive (required)") - staging = fs.String("d", "./staging", "staging directory for files in flight") - maxStaging = fs.String("m", "", "cap staging at this size, e.g. 40G (default: no cap)") - threads = fs.Int("threads", 4, "connections per file") - limit = fs.Int("limit", 2, "files downloading at once") - uploads = fs.Int("uploads", 2, "files uploading at once") - minFree = fs.Int64("min-free", 5, "stop if the remote has fewer than this many GiB free") - limitItems = fs.Int("limit-items", 0, "stop after this many files (0 means no limit)") - confirm = fs.Bool("confirm", true, "re-state each uploaded file to prove its size") - takeout = fs.Bool("takeout", true, "use a takeout session, as `tdl dl --takeout` did") - ns = fs.String("n", "default", "tdl session namespace") - dataDir = fs.String("storage", tdlkv.DefaultDir(), "tdl bolt storage directory") + chat = flags.String("c", "", "chat id, username, or t.me link (required)") + remoteArg = flags.String("r", "", "rclone destination, e.g. pikpak:archive (required)") + staging = flags.String("d", "./staging", "staging directory for files in flight") + maxStaging = flags.String("m", "", "cap staging at this size, e.g. 40G (default: no cap)") + threads = flags.Int("threads", 4, "connections per file") + limit = flags.Int("limit", 2, "files downloading at once") + uploads = flags.Int("uploads", 2, "files uploading at once") + minFree = flags.Int64("min-free", 5, "stop if the remote has fewer than this many GiB free") + limitItems = flags.Int("limit-items", 0, "stop after this many files (0 means no limit)") + confirm = flags.Bool("confirm", true, "re-state each uploaded file to prove its size") + takeout = flags.Bool("takeout", true, "use a takeout session, as `tdl dl --takeout` did") + ns = flags.String("n", "default", "tdl session namespace") + dataDir = flags.String("storage", tdlkv.DefaultDir(), "tdl bolt storage directory") ) for name, replacement := range retiredFlags { - fs.Var(retiredFlag{name, replacement}, name, "retired") + flags.Var(retiredFlag{name, replacement}, name, "retired") } - if err := fs.Parse(args); err != nil { + if err := flags.Parse(args); err != nil { if errors.Is(err, flag.ErrHelp) { return err } @@ -169,7 +169,11 @@ func syncCmd(ctx context.Context, args []string) error { fmt.Fprintf(os.Stderr, " message %d failed: %v\n", f.Item.MessageID, f.Err) } if runErr != nil { - return runErr + // Reported, not returned yet: a run that archived thousands of files + // and hit one transient upload error has still made progress, and + // suppressing the report would leave the operator — and any driver + // reading the exit code — unable to tell that from a total failure. + fmt.Fprintf(os.Stderr, "run ended early: %v\n", runErr) } // The remote is re-indexed rather than assumed: the run's own view of diff --git a/go.mod b/go.mod index c4cae8f..ee2e563 100644 --- a/go.mod +++ b/go.mod @@ -7,6 +7,7 @@ require ( github.com/iyear/tdl/core v0.20.4 github.com/rclone/rclone v1.75.1 go.etcd.io/bbolt v1.5.0 + golang.org/x/sync v0.22.0 ) require ( @@ -227,7 +228,6 @@ require ( golang.org/x/mod v0.38.0 // indirect golang.org/x/net v0.58.0 // indirect golang.org/x/oauth2 v0.36.0 // indirect - golang.org/x/sync v0.22.0 // indirect golang.org/x/sys v0.47.0 // indirect golang.org/x/term v0.45.0 // indirect golang.org/x/text v0.41.0 // indirect diff --git a/internal/pipeline/download.go b/internal/pipeline/download.go index e160c0a..5e4a0ba 100644 --- a/internal/pipeline/download.go +++ b/internal/pipeline/download.go @@ -8,6 +8,7 @@ import ( "os" "path/filepath" "strings" + "sync/atomic" "github.com/iyear/tdl/core/dcpool" "github.com/iyear/tdl/core/downloader" @@ -29,8 +30,14 @@ type DownloadOptions struct { Report func(Stats) // acquire reserves staging space before a download starts, blocking until - // there is room. Unset means no bound. + // there is room. Unset means no bound. release hands a reservation back for + // an item that never reaches a download. acquire func(context.Context, int64) error + release func(int64) + + // stop, when set, ends iteration cleanly from another goroutine — used to + // halt downloads once the destination has stopped accepting uploads. + stop *atomic.Bool // onReady hands a completed file to the upload leg; onFailed says nothing // was staged, so whatever acquire reserved must be given back. onReady func(tgsource.Item) @@ -39,12 +46,13 @@ type DownloadOptions struct { // Download fetches every item in seq into the staging directory. // -// Each file is written to .part and renamed to only once the -// downloader reports it complete, so a name without the suffix is always a -// whole file. That is what lets the upload half treat "the file exists" as -// "the file is finished" — the property the shell pipeline had to approximate -// with a filename convention plus an age guard, because it could not see -// inside tdl. +// Each file is written to .part and renamed to only once its size +// matches what Telegram reported, so a name without the suffix is always a whole +// file. Uploads are driven by completion rather than by scanning for that, but +// the invariant still matters: it is what makes a leftover file from an +// interrupted run safe to keep and a leftover .part safe to delete. The shell +// pipeline could only approximate it with a filename convention plus an age +// guard, because it could not see inside tdl. // // A failed item does not abort the run: it is recorded in the returned outcomes // and the rest continue, matching what a partial `tdl dl` pass did. @@ -61,6 +69,10 @@ func Download(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Downlo it := newElemIter(seq, o.Staging, o.Takeout) it.acquire = o.acquire + it.release = o.release + if o.stop != nil { + it.stopped = o.stop + } defer func() { _ = it.Close() }() prog := newProgress(func(e *elem, err error) error { @@ -85,6 +97,12 @@ func Download(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Downlo }).Download(ctx, o.Limit) outcomes, stats := prog.results() + // The iterator's failure is read only now, after Download has joined every + // worker. Reporting it through Iter.Err would have made Download skip that + // join entirely. + if err == nil { + err = it.failure + } return outcomes, stats, err } diff --git a/internal/pipeline/download_test.go b/internal/pipeline/download_test.go index 7940c46..9952a5b 100644 --- a/internal/pipeline/download_test.go +++ b/internal/pipeline/download_test.go @@ -1,6 +1,7 @@ package pipeline import ( + "context" "errors" "os" "path/filepath" @@ -141,12 +142,11 @@ func TestElemIterRejectsUnsafeNames(t *testing.T) { if it.Next(t.Context()) { t.Fatal("iterator accepted a name that escapes the staging directory") } - err := it.Err() - if err == nil { - t.Fatal("Err() = nil after rejecting an unsafe name") + if it.failure == nil { + t.Fatal("no failure recorded after rejecting an unsafe name") } - if !strings.Contains(err.Error(), "message 7") { - t.Errorf("error should name the message, got: %v", err) + if !strings.Contains(it.failure.Error(), "message 7") { + t.Errorf("failure should name the message, got: %v", it.failure) } } @@ -183,8 +183,8 @@ func TestElemIterOpensPartFilesAndPropagatesWalkErrors(t *testing.T) { if it.Next(t.Context()) { t.Fatal("Next() = true after a walk error") } - if !errors.Is(it.Err(), want) { - t.Errorf("Err() = %v, want %v", it.Err(), want) + if !errors.Is(it.failure, want) { + t.Errorf("failure = %v, want %v", it.failure, want) } }) } @@ -288,3 +288,112 @@ func TestFinishAcceptsExactSize(t *testing.T) { t.Errorf("final size = %d, want 2048", info.Size()) } } + +// Err must always report nil, however badly iteration went. +// +// core's Download skips wg.Wait entirely when Iter.Err is non-nil +// (downloader.go:65-68), returning while its workers are still running. The +// pipeline closes its upload channel as soon as Download returns, so a non-nil +// Err here means workers send on a closed channel and the process panics — +// on every Ctrl-C, since cancellation is one of the ways iteration stops. +func TestElemIterNeverReportsErrToTheDownloader(t *testing.T) { + staging := t.TempDir() + + cases := map[string]func() *elemIter{ + "walk error": func() *elemIter { + seq := func(yield func(tgsource.Item, error) bool) { + yield(tgsource.Item{}, errors.New("boom")) + } + return newElemIter(seq, staging, false) + }, + "unsafe name": func() *elemIter { + seq := func(yield func(tgsource.Item, error) bool) { + yield(testItem(t, 1, "../escape", 10), nil) + } + return newElemIter(seq, staging, false) + }, + "cancelled": func() *elemIter { + seq := func(yield func(tgsource.Item, error) bool) { + yield(testItem(t, 2, "a.mp4", 10), nil) + } + return newElemIter(seq, staging, false) + }, + } + + for name, build := range cases { + t.Run(name, func(t *testing.T) { + it := build() + defer func() { _ = it.Close() }() + + ctx := t.Context() + if name == "cancelled" { + cancelled, cancel := context.WithCancel(ctx) + cancel() + ctx = cancelled + } + + for it.Next(ctx) { + } + if err := it.Err(); err != nil { + t.Errorf("Err() = %v, want nil — a non-nil Err makes Download abandon its workers", err) + } + if it.failure == nil { + t.Error("the real failure was not stashed") + } + }) + } +} + +// A caller-set stop flag ends iteration without looking like a failure, which is +// how the circuit breaker halts downloads. +func TestElemIterStopsOnFlagWithoutRecordingFailure(t *testing.T) { + staging := t.TempDir() + seq := func(yield func(tgsource.Item, error) bool) { + for i := 1; i <= 5; i++ { + if !yield(testItem(t, i, "a.mp4", 10), nil) { + return + } + } + } + it := newElemIter(seq, staging, false) + defer func() { _ = it.Close() }() + + if !it.Next(t.Context()) { + t.Fatal("first Next() = false") + } + it.stopped.Store(true) + + if it.Next(t.Context()) { + t.Error("Next() = true after the stop flag was set") + } + if it.failure != nil { + t.Errorf("failure = %v, want nil — stopping is not a failure", it.failure) + } +} + +// A reservation must come back when the item never reaches a download, or the +// budget shrinks by that much for the rest of the run. +func TestElemIterReturnsReservationWhenOpenFails(t *testing.T) { + // A staging path that is a file, not a directory, makes OpenFile fail. + staging := filepath.Join(t.TempDir(), "not-a-dir") + if err := os.WriteFile(staging, []byte("x"), 0o600); err != nil { + t.Fatalf("seed: %v", err) + } + + seq := func(yield func(tgsource.Item, error) bool) { + yield(testItem(t, 1, "a.mp4", 4096), nil) + } + it := newElemIter(seq, staging, false) + defer func() { _ = it.Close() }() + + var acquired, released int64 + it.acquire = func(_ context.Context, n int64) error { acquired += n; return nil } + it.release = func(n int64) { released += n } + + if it.Next(t.Context()) { + t.Fatal("Next() succeeded with an unusable staging directory") + } + if acquired != released { + t.Errorf("acquired %d bytes but released %d — the reservation leaked", acquired, released) + } +} diff --git a/internal/pipeline/elem.go b/internal/pipeline/elem.go index d3c3464..703abe8 100644 --- a/internal/pipeline/elem.go +++ b/internal/pipeline/elem.go @@ -8,6 +8,7 @@ import ( "iter" "os" "path/filepath" + "sync/atomic" "github.com/gotd/td/tg" @@ -59,37 +60,57 @@ func finalPath(staging string, it tgsource.Item) string { // bridges them without this code owning a goroutine or a channel, which is why // Walk returns a sequence in the first place. type elemIter struct { - next func() (tgsource.Item, error, bool) - stop func() - staging string - takeout bool + next func() (tgsource.Item, error, bool) + stopPull func() + staging string + takeout bool // acquire reserves staging space for the next item. Blocking here is what // makes backpressure work: core's Download calls Next from its dispatch - // loop, so a blocked Next stops new downloads starting without stopping the - // uploads that free the space. + // loop (downloader.go:38), so a blocked Next stops new downloads starting + // without stopping the uploads that free the space. acquire func(context.Context, int64) error + // release hands a reservation back when the item never reaches a download. + release func(int64) current *elem - err error - // opened records every file handle so a run can close them all. The - // downloader never closes what To() hands it, and a leak here is thousands - // of descriptors on a full archive run. + // failure holds why iteration stopped, and Err deliberately does not return + // it. core's Download skips wg.Wait entirely when Iter.Err is non-nil + // (downloader.go:65-68), abandoning workers that are still running — which + // would let this package tear down its upload channel underneath them. So + // Next reports "no more items" and the caller reads failure() afterwards, + // guaranteeing every worker has finished first. + failure error + // stopped ends iteration without an error, for a caller that has decided the + // run cannot usefully continue. Supplied by the caller so it can be set from + // another goroutine without racing on the iterator itself. + stopped *atomic.Bool + + // opened records every file handle. finish closes each one on the normal + // path, so this is not what keeps descriptors from leaking; it is the + // backstop for items that were opened but never reached finish, which is + // what an aborted iteration leaves behind. opened []*os.File } func newElemIter(seq iter.Seq2[tgsource.Item, error], staging string, takeout bool) *elemIter { - next, stop := iter.Pull2(seq) - return &elemIter{next: next, stop: stop, staging: staging, takeout: takeout} + next, stopPull := iter.Pull2(seq) + return &elemIter{ + next: next, + stopPull: stopPull, + staging: staging, + takeout: takeout, + stopped: new(atomic.Bool), + } } func (i *elemIter) Next(ctx context.Context) bool { - if i.err != nil { + if i.failure != nil || i.stopped.Load() { return false } if err := ctx.Err(); err != nil { - i.err = err + i.failure = err return false } @@ -98,7 +119,7 @@ func (i *elemIter) Next(ctx context.Context) bool { return false } if err != nil { - i.err = err + i.failure = err return false } @@ -106,20 +127,25 @@ func (i *elemIter) Next(ctx context.Context) bool { // os.Create: the error names the message, and the run continues instead of // failing on a path that could never have worked. if err := naming.Safe(item.Name); err != nil { - i.err = fmt.Errorf("message %d: %w", item.MessageID, err) + i.failure = fmt.Errorf("message %d: %w", item.MessageID, err) return false } if i.acquire != nil { if err := i.acquire(ctx, item.Size()); err != nil { - i.err = err + i.failure = err return false } } f, err := os.OpenFile(partPath(i.staging, item), os.O_CREATE|os.O_RDWR, 0o600) if err != nil { - i.err = fmt.Errorf("open destination for message %d: %w", item.MessageID, err) + // The reservation is handed back here because this item will never + // reach a download, so no OnDone will ever release it for us. + if i.release != nil { + i.release(item.Size()) + } + i.failure = fmt.Errorf("open destination for message %d: %w", item.MessageID, err) return false } i.opened = append(i.opened, f) @@ -129,11 +155,14 @@ func (i *elemIter) Next(ctx context.Context) bool { } func (i *elemIter) Value() downloader.Elem { return i.current } -func (i *elemIter) Err() error { return i.err } + +// Err always reports nil so core's Download reaches wg.Wait and joins its +// workers. See the failure field. +func (i *elemIter) Err() error { return nil } // Close releases the pull iterator and every file the walk opened. func (i *elemIter) Close() error { - i.stop() + i.stopPull() var firstErr error for _, f := range i.opened { if err := f.Close(); err != nil && firstErr == nil { diff --git a/internal/pipeline/pipeline.go b/internal/pipeline/pipeline.go index 2bcd1b4..af9b391 100644 --- a/internal/pipeline/pipeline.go +++ b/internal/pipeline/pipeline.go @@ -5,7 +5,10 @@ import ( "errors" "fmt" "iter" + "os" + "path/filepath" "sync" + "sync/atomic" "github.com/rclone/rclone/fs" "golang.org/x/sync/semaphore" @@ -53,6 +56,11 @@ func (r Result) Failed() []Outcome { return out } +// maxRecordedErrors bounds what a run keeps from a failing remote. Past this, +// the pattern is established and joining thousands of identical strings just +// makes the final message unreadable. +const maxRecordedErrors = 10 + // Run downloads every item and uploads each one as it completes. // // This is the whole reason for the rewrite. run.sh could not see inside tdl, so @@ -60,13 +68,17 @@ func (r Result) Failed() []Outcome { // `du -sk` every ten seconds, and enforced its disk cap by sending SIGSTOP and // SIGCONT to the tdl process. None of that exists here. Completion is a function // returning. The cap is a semaphore: a download acquires its own size before -// starting and releases it only once the upload has confirmed, so when the -// remote is slow the acquire blocks and downloads pause on their own. +// starting and releases it once the file is off local disk, so when the remote +// is slow the acquire blocks and downloads pause on their own. // -// Blocking in the iterator is safe by construction — core's Download calls -// Iter.Next from its dispatch loop while workers run in an errgroup, so a -// blocked Next stalls new work without stopping the uploads that free the budget -// (downloader.go:36-63). +// Blocking in the iterator is safe, but not for the reason it first appears. +// core's Download calls Iter.Next from its dispatch loop while workers run in an +// errgroup, so a blocked Next stalls new work without stopping the uploads that +// free the budget. What is *not* safe is reporting an error through Iter.Err: +// Download then returns without joining its workers (downloader.go:65-68), and +// tearing down the upload channel underneath them panics. So elemIter always +// reports a nil Err and stashes the real one, which Download's return +// guarantees is safe to read. func Run(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Options) (Result, error) { if o.Uploads <= 0 { o.Uploads = 1 @@ -84,37 +96,51 @@ func Run(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Options) (R budget := newBudget(o.Budget) uploads := make(chan tgsource.Item, o.Uploads) - // Upload workers own the release side of the budget, so every path out of - // one — success, failure, cancellation — must release, or the run deadlocks - // with downloads waiting on space that is never freed. var ( - wg sync.WaitGroup - mu sync.Mutex - uploadErrs []error - streak int - tripped bool + wg sync.WaitGroup + mu sync.Mutex + errs []error + nErrs int + streak int + tripped bool ) - upCtx, tripRun := context.WithCancel(ctx) - defer tripRun() + + // stopDownloads ends the download side once the destination has stopped + // accepting work. It stops the iterator rather than cancelling a context, + // because cancelling only the uploads would leave downloads running at full + // speed against a remote that is refusing them — every file staying on disk, + // every reservation released on the way out. A broken remote would fill the + // local disk faster than a working one does. + stopDownloads := new(atomic.Bool) for range o.Uploads { wg.Add(1) go func() { defer wg.Done() for it := range uploads { - err := up.upload(upCtx, it) + err := up.upload(ctx, it) + + if err != nil { + // MoveFile leaves the local copy in place when it fails, so + // the reservation cannot simply be handed back — the bytes + // are still on disk. Removing the file first is what keeps + // the cap honest. + if rerr := os.Remove(filepath.Join(o.Staging, it.Name)); rerr != nil && !os.IsNotExist(rerr) { + err = errors.Join(err, fmt.Errorf("and it is still in staging: %w", rerr)) + } + } budget.release(it.Size()) mu.Lock() if err != nil { - uploadErrs = append(uploadErrs, err) + if nErrs < maxRecordedErrors { + errs = append(errs, err) + } + nErrs++ streak++ if streak >= o.MaxFailures && !tripped { - // A remote that fails this many times running is not - // going to recover on its own, and continuing just fills - // staging until the disk does. tripped = true - tripRun() + stopDownloads.Store(true) } } else { streak = 0 @@ -132,20 +158,24 @@ func Run(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Options) (R Takeout: o.Takeout, Report: o.Report, acquire: budget.acquire, + release: budget.release, onReady: func(it tgsource.Item) { uploads <- it }, onFailed: func(it tgsource.Item) { // Nothing was staged, so the reservation has to come back here // instead of from an upload that will never happen. budget.release(it.Size()) }, + stop: stopDownloads, }) + // Safe only because Download joined its workers, which is guaranteed by + // elemIter.Err always being nil. close(uploads) wg.Wait() mu.Lock() - errs := append([]error(nil), uploadErrs...) - trip := tripped + joined := errors.Join(errs...) + trip, total := tripped, nErrs mu.Unlock() res := Result{Stats: stats, Outcomes: dlOutcomes} @@ -153,10 +183,10 @@ func Run(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Options) (R case dlErr != nil: return res, dlErr case trip: - return res, fmt.Errorf("stopping after %d consecutive upload failures: %w", - o.MaxFailures, errors.Join(errs...)) - case len(errs) > 0: - return res, errors.Join(errs...) + return res, fmt.Errorf("stopped after %d consecutive upload failures (%d total): %w", + o.MaxFailures, total, joined) + case total > 0: + return res, fmt.Errorf("%d upload(s) failed: %w", total, joined) } return res, nil } @@ -176,10 +206,10 @@ func (b *budget) acquire(ctx context.Context, n int64) error { if b.sem == nil { return nil } - // An item larger than the whole budget could never be admitted and would - // block forever, so it is refused with an error that says what to change. - // Callers validate up front too; this is the guard for an item whose size - // was not known then. + // An item larger than the budget cannot be admitted, and semaphore.Acquire + // handles that by blocking until the context is cancelled rather than + // failing — so there is no error to surface and no guard to add here. The + // real protection is validateBudget refusing such a run before it starts. if err := b.sem.Acquire(ctx, n); err != nil { return fmt.Errorf("waiting for %d bytes of staging space: %w", n, err) } diff --git a/internal/pipeline/upload.go b/internal/pipeline/upload.go index 89ed3eb..f38f108 100644 --- a/internal/pipeline/upload.go +++ b/internal/pipeline/upload.go @@ -3,6 +3,7 @@ package pipeline import ( "context" "fmt" + "time" "github.com/rclone/rclone/fs" "github.com/rclone/rclone/fs/operations" @@ -21,13 +22,15 @@ type uploader struct { // arrived at the expected size. // // MoveFile removes the local copy as part of the move, so a successful return -// means the file is on the remote and off local disk — which is what lets the -// byte budget be released. +// means the file is on the remote and off local disk. // -// Confirmation closes a gap the shell pipeline left open: there, a truncated -// upload was only noticed by a later verify pass, after the local copy was -// already gone. Re-stating the object costs one round trip per file and turns a -// silent corruption into a retry. +// A short object is deleted rather than left in place, and that is the part that +// matters. Verification matches on name and non-zero size, so a truncated object +// under the right name would be counted archived by this run and by every run +// after it — permanently, with the local copy already gone. Removing it turns a +// silent corruption into an absent file the next run fetches again. The shell +// pipeline had this hole too: it noticed a bad upload only at the next verify, +// by which point the evidence was the same. func (u *uploader) upload(ctx context.Context, it tgsource.Item) error { if err := operations.MoveFile(ctx, u.dst, u.local, it.Name, it.Name); err != nil { return fmt.Errorf("move %q to %s: %w", it.Name, u.dst.String(), err) @@ -41,7 +44,17 @@ func (u *uploader) upload(ctx context.Context, it tgsource.Item) error { return fmt.Errorf("confirm %q: %w", it.Name, err) } if got := obj.Size(); got != it.Size() { - return fmt.Errorf("confirm %q: remote has %d bytes, expected %d", it.Name, got, it.Size()) + err := fmt.Errorf("confirm %q: remote has %d bytes, expected %d", it.Name, got, it.Size()) + // Deleted on a fresh context: the run may already be shutting down, and + // leaving a plausible-looking short object behind is worse than the + // error that got us here. + delCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), 30*time.Second) + defer cancel() + if derr := operations.DeleteFile(delCtx, obj); derr != nil { + return fmt.Errorf("%w (and it could not be removed: %v — delete it by hand "+ + "or verify will count it archived)", err, derr) + } + return err } return nil } diff --git a/internal/report/progress.go b/internal/report/progress.go index e458fa8..69cf2c7 100644 --- a/internal/report/progress.go +++ b/internal/report/progress.go @@ -41,10 +41,17 @@ func New(w io.Writer, total int, totalBytes int64) *Reporter { return &Reporter{w: w, tty: isTerminal(w), total: total, totalBytes: totalBytes, started: time.Now()} } -// Update renders a snapshot. Safe to call from several goroutines, and cheap -// enough to call on every progress callback. +// Update renders a snapshot. Safe to call from several goroutines. +// +// A contended update is dropped rather than queued. Every download worker calls +// this on each progress callback, so holding the lock across the write would +// make terminal latency — an ssh session with a slow link, say — throttle the +// downloads themselves. A skipped frame costs nothing; the next callback is +// milliseconds away and Finish always prints. func (r *Reporter) Update(s pipeline.Stats) { - r.mu.Lock() + if !r.mu.TryLock() { + return + } defer r.mu.Unlock() now := time.Now() From 3e75138c59d0ec8d86f23e9d8c7767e19a21a4a7 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Sun, 6 Sep 2026 19:51:24 +0700 Subject: [PATCH 06/16] fix: clean up a fragment left by a failed upload, and skip unwritable names MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit rclone writes straight to the final remote name on any backend that does not advertise PartialUploads, and cleans up after a failed Put only when it did not. Pikpak advertises neither, so a transfer that died halfway left a fragment under exactly the name verification matches on — counted archived by that run and every run after it, with the local copy already deleted. A failed move now looks for that object and removes it, leaving a complete one alone since pikpak's async commit can still land it correctly. An unwritable filename ended the whole walk, so one hostile name could strand every message behind it. It is skipped and reported instead. The comment there had described that behaviour all along. Also: remove the part file when promoting it fails, since the caller hands the reservation back and the cap would stay over-committed; report upload failures alongside a download error rather than instead of it, which on Ctrl-C hid that finished files had been discarded. Run had no test of its own because it called Download directly. That step is now indirected, covering the properties only the composition has: uploads closed after the last send, the budget balanced across failures, and a tripped breaker halting downloads rather than walking the whole chat. --- internal/pipeline/download.go | 12 +- internal/pipeline/download_test.go | 62 ++++-- internal/pipeline/elem.go | 59 ++++-- internal/pipeline/pipeline.go | 26 ++- internal/pipeline/run_test.go | 306 +++++++++++++++++++++++++++++ internal/pipeline/upload.go | 67 +++++-- internal/pipeline/upload_test.go | 68 +++++++ 7 files changed, 542 insertions(+), 58 deletions(-) create mode 100644 internal/pipeline/run_test.go create mode 100644 internal/pipeline/upload_test.go diff --git a/internal/pipeline/download.go b/internal/pipeline/download.go index 5e4a0ba..7d9bf8a 100644 --- a/internal/pipeline/download.go +++ b/internal/pipeline/download.go @@ -103,7 +103,10 @@ func Download(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Downlo if err == nil { err = it.failure } - return outcomes, stats, err + // Skipped items are reported alongside whatever else happened rather than + // instead of it: the run did real work, and the caller still needs to know + // these messages were never attempted. + return outcomes, stats, errors.Join(append([]error{err}, it.skipped...)...) } // finish closes a downloaded file and either promotes it or removes it. @@ -143,6 +146,13 @@ func finish(staging string, e *elem, downloadErr error) error { } if err := os.Rename(part, finalPath(staging, e.item)); err != nil { + // The part file goes too. The caller treats this as a failure and hands + // the byte reservation back, so leaving the file on disk would put the + // staging cap permanently over-committed by its size. + if rerr := os.Remove(part); rerr != nil && !os.IsNotExist(rerr) { + return errors.Join(fmt.Errorf("promote %q: %w", part, err), + fmt.Errorf("and it is still in staging: %w", rerr)) + } return fmt.Errorf("promote %q: %w", part, err) } return nil diff --git a/internal/pipeline/download_test.go b/internal/pipeline/download_test.go index 9952a5b..691fbdb 100644 --- a/internal/pipeline/download_test.go +++ b/internal/pipeline/download_test.go @@ -129,9 +129,47 @@ func TestSweepPartialsOnMissingDirectory(t *testing.T) { } } -// An unwritable name must stop the iterator with a message naming the message, -// rather than surfacing as a bare os.Create failure later. -func TestElemIterRejectsUnsafeNames(t *testing.T) { +// An unwritable name must be refused with a message naming the message, rather +// than surfacing as a bare os.Create failure later — and it must be skipped, not +// treated as the end of the walk. One hostile filename cannot be allowed to +// strand every message behind it. +func TestElemIterSkipsUnsafeNamesAndKeepsGoing(t *testing.T) { + staging := t.TempDir() + bad := testItem(t, 7, "../../escape.conf", 10) + good := testItem(t, 8, "fine.mp4", 10) + + seq := func(yield func(tgsource.Item, error) bool) { + if !yield(bad, nil) { + return + } + yield(good, nil) + } + it := newElemIter(seq, staging, false) + defer func() { _ = it.Close() }() + + if !it.Next(t.Context()) { + t.Fatal("an unsafe name ended the walk; the item after it was never reached") + } + if got := it.current.item.MessageID; got != 8 { + t.Fatalf("Next yielded message %d, want the item after the unsafe one", got) + } + if it.Next(t.Context()) { + t.Error("iterator produced a third item") + } + if it.failure != nil { + t.Errorf("a skipped item must not fail the run, got: %v", it.failure) + } + if len(it.skipped) != 1 { + t.Fatalf("skipped = %d, want 1", len(it.skipped)) + } + if !strings.Contains(it.skipped[0].Error(), "message 7") { + t.Errorf("the skip should name the message, got: %v", it.skipped[0]) + } +} + +// The skipped items still have to reach the caller: the run did work, but these +// messages were never attempted and nothing else would say so. +func TestDownloadReportsSkippedItems(t *testing.T) { staging := t.TempDir() bad := testItem(t, 7, "../../escape.conf", 10) @@ -139,14 +177,14 @@ func TestElemIterRejectsUnsafeNames(t *testing.T) { it := newElemIter(seq, staging, false) defer func() { _ = it.Close() }() - if it.Next(t.Context()) { - t.Fatal("iterator accepted a name that escapes the staging directory") + for it.Next(t.Context()) { } - if it.failure == nil { - t.Fatal("no failure recorded after rejecting an unsafe name") + err := errors.Join(append([]error{nil}, it.skipped...)...) + if err == nil { + t.Fatal("skipped items produced no error for the caller") } - if !strings.Contains(it.failure.Error(), "message 7") { - t.Errorf("failure should name the message, got: %v", it.failure) + if !strings.Contains(err.Error(), "message 7") { + t.Errorf("error should name the skipped message, got: %v", err) } } @@ -306,12 +344,6 @@ func TestElemIterNeverReportsErrToTheDownloader(t *testing.T) { } return newElemIter(seq, staging, false) }, - "unsafe name": func() *elemIter { - seq := func(yield func(tgsource.Item, error) bool) { - yield(testItem(t, 1, "../escape", 10), nil) - } - return newElemIter(seq, staging, false) - }, "cancelled": func() *elemIter { seq := func(yield func(tgsource.Item, error) bool) { yield(testItem(t, 2, "a.mp4", 10), nil) diff --git a/internal/pipeline/elem.go b/internal/pipeline/elem.go index 703abe8..4738202 100644 --- a/internal/pipeline/elem.go +++ b/internal/pipeline/elem.go @@ -87,6 +87,11 @@ type elemIter struct { // another goroutine without racing on the iterator itself. stopped *atomic.Bool + // skipped collects items refused before any download was attempted. They do + // not stop the run: one message with a hostile filename must not be able to + // strand every message behind it, which is what ending iteration would mean. + skipped []error + // opened records every file handle. finish closes each one on the normal // path, so this is not what keeps descriptors from leaking; it is the // backstop for items that were opened but never reached finish, which is @@ -106,29 +111,41 @@ func newElemIter(seq iter.Seq2[tgsource.Item, error], staging string, takeout bo } func (i *elemIter) Next(ctx context.Context) bool { - if i.failure != nil || i.stopped.Load() { - return false - } - if err := ctx.Err(); err != nil { - i.failure = err - return false - } + var item tgsource.Item + for { + if i.failure != nil || i.stopped.Load() { + return false + } + if err := ctx.Err(); err != nil { + i.failure = err + return false + } - item, err, ok := i.next() - if !ok { - return false - } - if err != nil { - i.failure = err - return false - } + var ( + err error + ok bool + ) + item, err, ok = i.next() + if !ok { + return false + } + if err != nil { + i.failure = err + return false + } - // A name that cannot be written is refused here rather than left to - // os.Create: the error names the message, and the run continues instead of - // failing on a path that could never have worked. - if err := naming.Safe(item.Name); err != nil { - i.failure = fmt.Errorf("message %d: %w", item.MessageID, err) - return false + // A name that cannot be written is refused here rather than left to + // os.Create, and refusing it skips the item rather than ending the walk. + // selectTodo already filters these out on the CLI path, so reaching this + // is either a second caller or a gap there; in both cases one unwritable + // name must not strand the rest of the chat behind it. + if err := naming.Safe(item.Name); err != nil { + if len(i.skipped) < maxRecordedErrors { + i.skipped = append(i.skipped, fmt.Errorf("message %d: %w", item.MessageID, err)) + } + continue + } + break } if i.acquire != nil { diff --git a/internal/pipeline/pipeline.go b/internal/pipeline/pipeline.go index af9b391..63a2487 100644 --- a/internal/pipeline/pipeline.go +++ b/internal/pipeline/pipeline.go @@ -61,6 +61,14 @@ func (r Result) Failed() []Outcome { // makes the final message unreadable. const maxRecordedErrors = 10 +// download is the download step, indirected so a test can drive Run's +// composition without a live Telegram connection. What that buys is coverage of +// the three properties Run alone is responsible for — that the upload channel is +// closed only after every send, that the byte budget balances across a whole +// run, and that a tripped breaker still terminates — none of which the pieces +// can be tested for individually. +var download = Download + // Run downloads every item and uploads each one as it completes. // // This is the whole reason for the rewrite. run.sh could not see inside tdl, so @@ -150,7 +158,7 @@ func Run(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Options) (R }() } - dlOutcomes, stats, dlErr := Download(ctx, seq, DownloadOptions{ + dlOutcomes, stats, dlErr := download(ctx, seq, DownloadOptions{ Pool: o.Pool, Staging: o.Staging, Threads: o.Threads, @@ -178,17 +186,21 @@ func Run(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Options) (R trip, total := tripped, nErrs mu.Unlock() - res := Result{Stats: stats, Outcomes: dlOutcomes} + // Both halves are reported. On the most common failure path — Ctrl-C — the + // download side returns context.Canceled while the upload workers drain + // whatever is still queued, fail every one of them against the cancelled + // context, and delete the staged file each time. Returning only the download + // error would leave the operator with "context canceled" and no sign that + // finished files had been discarded. + var upErr error switch { - case dlErr != nil: - return res, dlErr case trip: - return res, fmt.Errorf("stopped after %d consecutive upload failures (%d total): %w", + upErr = fmt.Errorf("stopped after %d consecutive upload failures (%d total): %w", o.MaxFailures, total, joined) case total > 0: - return res, fmt.Errorf("%d upload(s) failed: %w", total, joined) + upErr = fmt.Errorf("%d upload(s) failed: %w", total, joined) } - return res, nil + return Result{Stats: stats, Outcomes: dlOutcomes}, errors.Join(dlErr, upErr) } // budget bounds how many bytes of downloaded-but-not-yet-uploaded data sit on diff --git a/internal/pipeline/run_test.go b/internal/pipeline/run_test.go new file mode 100644 index 0000000..e0841e3 --- /dev/null +++ b/internal/pipeline/run_test.go @@ -0,0 +1,306 @@ +package pipeline + +import ( + "context" + "errors" + "fmt" + "iter" + "os" + "path/filepath" + "strings" + "sync/atomic" + "testing" + + _ "github.com/rclone/rclone/backend/local" + "github.com/rclone/rclone/fs" + + "github.com/iyear/tdl/core/tmedia" + + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" +) + +// These exercise Run's composition rather than its parts. Everything underneath +// it is tested directly, but the properties Run alone owns — that the upload +// channel closes only after the last send, that reservations balance across a +// whole run, that a tripped breaker still terminates — only exist once the +// pieces are wired together, and a regression in any of them is silent. + +// runFake substitutes the download step for the duration of a test. +func runFake(t *testing.T, f func(context.Context, iter.Seq2[tgsource.Item, error], DownloadOptions) ([]Outcome, Stats, error)) { + t.Helper() + prev := download + download = f + t.Cleanup(func() { download = prev }) +} + +// stageItems is a download step that writes each item's bytes into staging and +// hands it to the upload leg, mimicking what the real one does on success. +// Items named in fail never reach staging. +func stageItems(fail map[int]bool) func(context.Context, iter.Seq2[tgsource.Item, error], DownloadOptions) ([]Outcome, Stats, error) { + return func(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o DownloadOptions) ([]Outcome, Stats, error) { + var ( + outcomes []Outcome + stats Stats + ) + for it, err := range seq { + if err != nil { + return outcomes, stats, err + } + if o.stop != nil && o.stop.Load() { + break + } + if o.acquire != nil { + if aerr := o.acquire(ctx, it.Size()); aerr != nil { + return outcomes, stats, aerr + } + } + stats.Started++ + if fail[it.MessageID] { + stats.Failed++ + outcomes = append(outcomes, Outcome{Item: it, Err: errors.New("download failed")}) + o.onFailed(it) + continue + } + if werr := os.WriteFile(filepath.Join(o.Staging, it.Name), + make([]byte, it.Size()), 0o600); werr != nil { + return outcomes, stats, werr + } + stats.Done++ + stats.BytesDone += it.Size() + outcomes = append(outcomes, Outcome{Item: it}) + o.onReady(it) + } + return outcomes, stats, nil + } +} + +func testItems(n int, size int64) []tgsource.Item { + out := make([]tgsource.Item, n) + for i := range out { + out[i] = tgsource.Item{ + MessageID: i + 1, + Name: fmt.Sprintf("-100123_%d_file.bin", i+1), + Media: &tmedia.Media{Size: size}, + } + } + return out +} + +func seqOf(items []tgsource.Item) iter.Seq2[tgsource.Item, error] { + return func(yield func(tgsource.Item, error) bool) { + for _, it := range items { + if !yield(it, nil) { + return + } + } + } +} + +// runOpts wires Run against local directories, so uploads are real rclone moves. +func runOpts(t *testing.T, budget int64) (Options, string, string) { + t.Helper() + staging, dstDir := t.TempDir(), t.TempDir() + dst, err := fs.NewFs(t.Context(), dstDir) + if err != nil { + t.Fatalf("open destination: %v", err) + } + return Options{ + Dst: dst, + Staging: staging, + Uploads: 2, + Budget: budget, + Confirm: true, + }, staging, dstDir +} + +func TestRunUploadsEveryDownloadedItem(t *testing.T) { + runFake(t, stageItems(nil)) + o, staging, dstDir := runOpts(t, 0) + items := testItems(6, 512) + + res, err := Run(t.Context(), seqOf(items), o) + if err != nil { + t.Fatalf("Run: %v", err) + } + if res.Stats.Done != len(items) { + t.Errorf("Done = %d, want %d", res.Stats.Done, len(items)) + } + for _, it := range items { + info, serr := os.Stat(filepath.Join(dstDir, it.Name)) + if serr != nil { + t.Errorf("%s not on the destination: %v", it.Name, serr) + continue + } + if info.Size() != it.Size() { + t.Errorf("%s is %d bytes, want %d", it.Name, info.Size(), it.Size()) + } + } + assertEmpty(t, staging) +} + +// The reason Err always reports nil: if Run closed the upload channel before +// the download step finished handing over items, this panics. +func TestRunClosesUploadsOnlyAfterTheLastSend(t *testing.T) { + var late atomic.Bool + runFake(t, func(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o DownloadOptions) ([]Outcome, Stats, error) { + // A worker still delivering after the step's own error is exactly what + // core does when it skips wg.Wait, so send one and then fail. + it := testItems(1, 128)[0] + if err := os.WriteFile(filepath.Join(o.Staging, it.Name), make([]byte, it.Size()), 0o600); err != nil { + return nil, Stats{}, err + } + o.acquire(ctx, it.Size()) + o.onReady(it) + late.Store(true) + return []Outcome{{Item: it}}, Stats{Started: 1, Done: 1}, errors.New("download step failed") + }) + o, staging, dstDir := runOpts(t, 0) + + _, err := Run(t.Context(), seqOf(nil), o) + if err == nil { + t.Fatal("Run returned nil, want the download step's error") + } + if !late.Load() { + t.Fatal("the download step never ran") + } + // The handed-over item must still have been uploaded, not dropped. + if _, serr := os.Stat(filepath.Join(dstDir, "-100123_1_file.bin")); serr != nil { + t.Errorf("item handed over before the error was not uploaded: %v", serr) + } + assertEmpty(t, staging) +} + +// A reservation that is not returned shrinks the cap for the rest of the run, +// and one returned twice panics. Neither is visible in the pieces individually. +// +// The budget here is exactly one file, so the run can only proceed if every +// reservation comes back: the second item cannot start until the first is +// released. A leak deadlocks and this test times out rather than passing +// quietly, which is the whole point of sizing it this way. +func TestRunBalancesTheBudgetAcrossFailures(t *testing.T) { + const size = 1024 + runFake(t, stageItems(map[int]bool{2: true, 5: true})) + o, staging, dstDir := runOpts(t, size) + items := testItems(8, size) + + res, err := Run(t.Context(), seqOf(items), o) + if err != nil { + t.Fatalf("Run: %v", err) + } + if got := len(res.Failed()); got != 2 { + t.Errorf("Failed() = %d, want 2", got) + } + if res.Stats.Done != 6 { + t.Errorf("Done = %d, want 6", res.Stats.Done) + } + // The six that downloaded are on the destination; the two that failed are + // not, and neither is holding space. + for _, it := range items { + _, serr := os.Stat(filepath.Join(dstDir, it.Name)) + wantThere := it.MessageID != 2 && it.MessageID != 5 + if wantThere && serr != nil { + t.Errorf("%s should be on the destination: %v", it.Name, serr) + } + if !wantThere && !os.IsNotExist(serr) { + t.Errorf("%s should not be on the destination", it.Name) + } + } + assertEmpty(t, staging) +} + +func TestRunStopsDownloadingAfterConsecutiveUploadFailures(t *testing.T) { + const size = 256 + items := testItems(40, size) + + var dispatched atomic.Int64 + runFake(t, func(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o DownloadOptions) ([]Outcome, Stats, error) { + var ( + outcomes []Outcome + stats Stats + ) + for it := range seqValues(seq) { + if o.stop.Load() { + break + } + o.acquire(ctx, it.Size()) + dispatched.Add(1) + // Nothing is written to staging, so every upload fails to find it. + stats.Started++ + stats.Done++ + outcomes = append(outcomes, Outcome{Item: it}) + o.onReady(it) + } + return outcomes, stats, nil + }) + o, staging, _ := runOpts(t, 0) + o.MaxFailures = 3 + + _, err := Run(t.Context(), seqOf(items), o) + if err == nil { + t.Fatal("Run returned nil, want the breaker's error") + } + if !strings.Contains(err.Error(), "consecutive upload failures") { + t.Errorf("error does not mention the breaker: %v", err) + } + // The point of stopping the iterator rather than cancelling uploads: the + // download side must not have walked the whole chat. + if got := dispatched.Load(); got == int64(len(items)) { + t.Errorf("all %d items were dispatched; the breaker did not stop downloads", got) + } + assertEmpty(t, staging) +} + +func seqValues(seq iter.Seq2[tgsource.Item, error]) iter.Seq[tgsource.Item] { + return func(yield func(tgsource.Item) bool) { + for it, err := range seq { + if err != nil { + return + } + if !yield(it) { + return + } + } + } +} + +// A failed upload must take the staged file with it, or the cap is over-committed +// by that much for the rest of the run. +func TestRunClearsStagingWhenAnUploadFails(t *testing.T) { + const size = 512 + runFake(t, func(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o DownloadOptions) ([]Outcome, Stats, error) { + it := testItems(1, size)[0] + // Staged at the wrong size, so the confirm step rejects it. + if err := os.WriteFile(filepath.Join(o.Staging, it.Name), make([]byte, size/2), 0o600); err != nil { + return nil, Stats{}, err + } + o.acquire(ctx, it.Size()) + o.onReady(it) + return []Outcome{{Item: it}}, Stats{Started: 1, Done: 1}, nil + }) + o, staging, dstDir := runOpts(t, 0) + + _, err := Run(t.Context(), seqOf(nil), o) + if err == nil { + t.Fatal("Run returned nil, want the confirm failure") + } + assertEmpty(t, staging) + // And the short object must not be left under the name verify matches. + if _, serr := os.Stat(filepath.Join(dstDir, "-100123_1_file.bin")); !os.IsNotExist(serr) { + t.Errorf("short object left on the destination: %v", serr) + } +} + +func assertEmpty(t *testing.T, dir string) { + t.Helper() + entries, err := os.ReadDir(dir) + if err != nil { + t.Fatalf("read staging: %v", err) + } + if len(entries) != 0 { + names := make([]string, len(entries)) + for i, e := range entries { + names[i] = e.Name() + } + t.Errorf("staging is not empty: %v", names) + } +} diff --git a/internal/pipeline/upload.go b/internal/pipeline/upload.go index f38f108..47670eb 100644 --- a/internal/pipeline/upload.go +++ b/internal/pipeline/upload.go @@ -2,6 +2,7 @@ package pipeline import ( "context" + "errors" "fmt" "time" @@ -24,16 +25,22 @@ type uploader struct { // MoveFile removes the local copy as part of the move, so a successful return // means the file is on the remote and off local disk. // -// A short object is deleted rather than left in place, and that is the part that -// matters. Verification matches on name and non-zero size, so a truncated object -// under the right name would be counted archived by this run and by every run -// after it — permanently, with the local copy already gone. Removing it turns a -// silent corruption into an absent file the next run fetches again. The shell -// pipeline had this hole too: it noticed a bad upload only at the next verify, -// by which point the evidence was the same. +// Both failure paths end at dropShort, and that is the part that matters. An +// object of the wrong size sitting under the right name is worse than no object +// at all: verification matches on name and non-zero size, so it would be counted +// archived by this run and by every run after it — permanently, once the local +// copy is gone. Removing it turns a silent corruption into an absent file the +// next run fetches again. func (u *uploader) upload(ctx context.Context, it tgsource.Item) error { if err := operations.MoveFile(ctx, u.dst, u.local, it.Name, it.Name); err != nil { - return fmt.Errorf("move %q to %s: %w", it.Name, u.dst.String(), err) + err = fmt.Errorf("move %q to %s: %w", it.Name, u.dst.String(), err) + // A failed move can still leave a partial object under the final name. + // rclone only writes to a temporary name when the backend advertises + // PartialUploads (copy.go:93), and it only cleans up after itself when + // it did (copy.go:348-350) — pikpak, the remote this was built against, + // advertises neither, so a died-halfway transfer stays exactly where a + // complete one would be. + return errors.Join(err, u.dropShort(ctx, it)) } if !u.confirm { return nil @@ -45,12 +52,7 @@ func (u *uploader) upload(ctx context.Context, it tgsource.Item) error { } if got := obj.Size(); got != it.Size() { err := fmt.Errorf("confirm %q: remote has %d bytes, expected %d", it.Name, got, it.Size()) - // Deleted on a fresh context: the run may already be shutting down, and - // leaving a plausible-looking short object behind is worse than the - // error that got us here. - delCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), 30*time.Second) - defer cancel() - if derr := operations.DeleteFile(delCtx, obj); derr != nil { + if derr := remove(ctx, obj); derr != nil { return fmt.Errorf("%w (and it could not be removed: %v — delete it by hand "+ "or verify will count it archived)", err, derr) } @@ -58,3 +60,40 @@ func (u *uploader) upload(ctx context.Context, it tgsource.Item) error { } return nil } + +// dropShort removes an object left under it.Name at the wrong size. +// +// An object of the *right* size is deliberately left alone. pikpak commits an +// upload as a server-side async task, so a transfer rclone gave up on can still +// land correctly afterwards; deleting it on the strength of the error alone +// would throw away a good file and force it to be fetched again. +func (u *uploader) dropShort(ctx context.Context, it tgsource.Item) error { + // A fresh context: the run may already be shutting down, which is one of + // the ways the move failed in the first place. + ctx, cancel := context.WithTimeout(context.WithoutCancel(ctx), 30*time.Second) + defer cancel() + + obj, err := u.dst.NewObject(ctx, it.Name) + if errors.Is(err, fs.ErrorObjectNotFound) { + return nil // nothing was left behind + } + if err != nil { + return fmt.Errorf("check for a leftover %q: %w", it.Name, err) + } + if obj.Size() == it.Size() { + return nil + } + if derr := remove(ctx, obj); derr != nil { + return fmt.Errorf("a %d-byte fragment of %q (expected %d) is on the remote and "+ + "could not be removed: %w — delete it by hand or verify will count it archived", + obj.Size(), it.Name, it.Size(), derr) + } + return nil +} + +// remove deletes an object on a context that outlives the run's cancellation. +func remove(ctx context.Context, obj fs.Object) error { + ctx, cancel := context.WithTimeout(context.WithoutCancel(ctx), 30*time.Second) + defer cancel() + return operations.DeleteFile(ctx, obj) +} diff --git a/internal/pipeline/upload_test.go b/internal/pipeline/upload_test.go new file mode 100644 index 0000000..c61f5ad --- /dev/null +++ b/internal/pipeline/upload_test.go @@ -0,0 +1,68 @@ +package pipeline + +import ( + "os" + "path/filepath" + "testing" + + _ "github.com/rclone/rclone/backend/local" + "github.com/rclone/rclone/fs" + + "github.com/iyear/tdl/core/tmedia" + + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" +) + +// dropShort is what stands between a died-halfway upload and a permanently +// wrong archive. rclone writes straight to the final name on any backend that +// does not advertise PartialUploads (copy.go:93) and cleans up only when it did +// not (copy.go:348-350), so on such a remote — pikpak, here — a fragment is left +// under exactly the name verification matches on. +func TestDropShortRemovesAFragmentButKeepsACompleteFile(t *testing.T) { + const name = "-100123_4242_clip.mp4" + const want = 4096 + + cases := map[string]struct { + staged int // bytes already at the destination; -1 means absent + wantThere bool + }{ + // A fragment must go: left alone, verify counts it archived by name and + // non-zero size, for this run and every run after it. + "fragment is removed": {staged: 400, wantThere: false}, + // A complete file must stay. pikpak commits uploads as a server-side + // async task, so a transfer rclone gave up on can still land correctly; + // deleting on the strength of the error alone throws away a good file. + "complete file is kept": {staged: want, wantThere: true}, + "nothing to clean up": {staged: -1, wantThere: false}, + } + + for label, tc := range cases { + t.Run(label, func(t *testing.T) { + dstDir := t.TempDir() + path := filepath.Join(dstDir, name) + if tc.staged >= 0 { + if err := os.WriteFile(path, make([]byte, tc.staged), 0o600); err != nil { + t.Fatalf("stage destination file: %v", err) + } + } + dst, err := fs.NewFs(t.Context(), dstDir) + if err != nil { + t.Fatalf("open destination: %v", err) + } + + u := &uploader{dst: dst} + it := tgsource.Item{MessageID: 4242, Name: name, Media: &tmedia.Media{Size: want}} + if err := u.dropShort(t.Context(), it); err != nil { + t.Fatalf("dropShort: %v", err) + } + + _, serr := os.Stat(path) + switch { + case tc.wantThere && serr != nil: + t.Errorf("a complete file was deleted: %v", serr) + case !tc.wantThere && serr == nil: + t.Error("a short object was left under the name verify matches") + } + }) + } +} From 19837ceb857d5e2f951b8e540923baafd8584305 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Sun, 6 Sep 2026 20:28:09 +0700 Subject: [PATCH 07/16] fix: reject names rclone rewrites, and stop resolving link markers as chats MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit rclone does not address a file by the bytes os.OpenFile wrote. Names handed to an Fs go through the backend encoder and names listed back are re-encoded to the standard set, neither of which the write path performs. A filename holding one of the rewritten characters was therefore stored under one string and looked up under another: confirmed against the local backend, where a written "a‛b.jpg" reports object not found and a written "a\nb.jpg" lists back as "a␊b.jpg". That is the same two-derivation divergence this program was written to remove, with rclone's encoder standing where filenamify used to. Such names are rejected, not encoded, for the same reason every other name is. The length limit ignored the ".part" suffix that is opened first, so a name just inside NAME_MAX passed the check and then failed to open on every pass, stalling the walk on that message forever. The suffix now lives beside the limit that has to account for it. filter.NewFilter(nil) does not build a neutral filter; it copies the package global, which rclone has already filled from RCLONE_*. Indexing inherited the operator's environment, so a stray RCLONE_MIN_SIZE emptied the index and re-downloaded the archive. Every narrowing field is now set explicitly and the result is asserted inactive. A subprocess test covers it, since the env is read at package init and t.Setenv is too late to observe anything. An ErrorDirNotFound from a subdirectory was also treated as an empty destination, returning a partial index as authoritative. t.me/c/ and t.me/s/ were passed through whole, and gotd reads the first path component as the username — resolving "c" or "s", which is a confusing failure at best and someone else's chat at worst, since one-character usernames exist. Both now yield the chat, and t.me/s// is refused like any other message link. Two tests asserted the old behaviour. core's dcpool.Takeout deadlocks when takeout init fails: it holds the pool mutex and recovers by calling Client, which takes the same non-reentrant mutex. Telegram returns TAKEOUT_INIT_DELAY for a takeout started recently and takeout is on by default, so two runs in succession hang the process with no output and no response to cancellation. The session is established once here instead, falling back to a plain client, and the pool's own Takeout is never called. --- .claude/agent-memory/code-reviewer/MEMORY.md | 4 + .../code-reviewer/rclone-embedding-traps.md | 36 ++++++++ .../code-reviewer/repo-exit-code-contract.md | 25 ++++++ internal/naming/safe.go | 38 +++++++- internal/naming/safe_test.go | 57 +++++++++++- internal/pipeline/elem.go | 6 +- internal/remote/index.go | 28 +++++- internal/remote/index_env_test.go | 64 ++++++++++++++ internal/tgsource/chat.go | 87 +++++++++++++++---- internal/tgsource/chat_test.go | 14 ++- internal/tgsource/client.go | 2 +- internal/tgsource/takeout.go | 62 +++++++++++++ internal/tgsource/takeout_test.go | 64 ++++++++++++++ 13 files changed, 459 insertions(+), 28 deletions(-) create mode 100644 .claude/agent-memory/code-reviewer/MEMORY.md create mode 100644 .claude/agent-memory/code-reviewer/rclone-embedding-traps.md create mode 100644 .claude/agent-memory/code-reviewer/repo-exit-code-contract.md create mode 100644 internal/remote/index_env_test.go create mode 100644 internal/tgsource/takeout.go create mode 100644 internal/tgsource/takeout_test.go diff --git a/.claude/agent-memory/code-reviewer/MEMORY.md b/.claude/agent-memory/code-reviewer/MEMORY.md new file mode 100644 index 0000000..7098a1a --- /dev/null +++ b/.claude/agent-memory/code-reviewer/MEMORY.md @@ -0,0 +1,4 @@ +# Agent Memory Index + +- [rclone embedding traps](rclone-embedding-traps.md) — `NewFilter(nil)` is not neutral; `RCLONE_*` reaches globals at package init, so `t.Setenv` tests prove nothing. +- [Exit-code contract](repo-exit-code-contract.md) — exit 1 is the driver's retry signal; a non-retryable state that exits 1 loops forever. diff --git a/.claude/agent-memory/code-reviewer/rclone-embedding-traps.md b/.claude/agent-memory/code-reviewer/rclone-embedding-traps.md new file mode 100644 index 0000000..5ebc642 --- /dev/null +++ b/.claude/agent-memory/code-reviewer/rclone-embedding-traps.md @@ -0,0 +1,36 @@ +--- +name: rclone-embedding-traps +description: Non-obvious rclone-as-a-library behaviours verified against rclone v1.75.1 source that keep biting this repo's remote/index code +metadata: + type: project +--- + +rclone v1.75.1 embedded as a library reads `RCLONE_*` environment variables into +package-level global option structs at **package init**, via +`fs.RegisterGlobalOptions` -> `OptionsInfo.load()` -> `fs.ConfigMap` -> +`optionEnvVars.Get` (`fs/registry.go:498-527`, `fs/configmap.go:115-152`). No +cobra/pflag wiring is needed for this to happen. + +Two consequences that are easy to get backwards, both verified empirically: + +- `filter.NewFilter(nil)` copies `filter.Opt` (`fs/filter/filter.go:196-202`), + which already holds the env-derived values. It is **not** a neutral filter. + `RCLONE_EXCLUDE` / `RCLONE_FILTER` / `RCLONE_MIN_SIZE` / `RCLONE_MAX_AGE` all + survive it. A genuinely neutral filter needs explicit + `&filter.Options{MinAge: fs.DurationOff, MaxAge: fs.DurationOff, MinSize: -1, MaxSize: -1}` + — the zero-value `filter.Options{}` errors with "min-age can't be larger than + max-age". +- `fs.AddConfig(ctx)` + assigning `ci.MaxDepth` *does* override + `RCLONE_MAX_DEPTH`, and env values for `Transfers` / `LowLevelRetries` really + are present in `fs.GetConfig(ctx)` without any flag parsing. + +**Why:** an index built through a narrowed listing does not fail — it reports +archived files as absent and re-downloads them, which is the exact multi-GiB +failure this project exists to remove. + +**How to apply:** when reviewing anything that lists a remote, check the filter +is neutralised explicitly rather than with `NewFilter(nil)`, and prove it with a +subprocess run under the env var rather than a `t.Setenv` test — `t.Setenv` runs +long after rclone's package init, so such a test passes regardless. + +Related: [[repo-exit-code-contract]] diff --git a/.claude/agent-memory/code-reviewer/repo-exit-code-contract.md b/.claude/agent-memory/code-reviewer/repo-exit-code-contract.md new file mode 100644 index 0000000..724fe7e --- /dev/null +++ b/.claude/agent-memory/code-reviewer/repo-exit-code-contract.md @@ -0,0 +1,25 @@ +--- +name: repo-exit-code-contract +description: tgexport's exit codes are a machine contract an external until-complete driver loops on; miscategorising one causes an infinite retry loop +metadata: + type: project +--- + +`tgexport` exit codes: 0 complete, 1 ran but files remain, 2 usage, 3 +remote/Telegram failure, 130/143 interrupted. Code 1 is the *retry* signal — an +external driver re-invokes on 1 and stops on 0 or 3. + +**Why:** the retired `export-until-complete.sh` had two stop conditions the Go +rewrite deliberately dropped (`plans/.../phase-06-cli-resume-and-observability.md` +lines 46-57): a pass counter, and a no-progress detector. The Go tool converges +in one invocation instead, so the exit code is now the *only* thing that can +stop a driver. + +**How to apply:** when reviewing any new error path in `cmd/tgexport`, ask which +exit code it produces and whether that state is actually retryable. Two states +have no representation in the contract and therefore fall through to 1 forever: +work that can never succeed (an unwritable filename, permanently unavailable +media) and a remote that is full or broken mid-run. Treat "exit 1 with no +possible progress" as a blocking defect, not a cosmetic one. + +Related: [[rclone-embedding-traps]] diff --git a/internal/naming/safe.go b/internal/naming/safe.go index feebf6f..a4e7c72 100644 --- a/internal/naming/safe.go +++ b/internal/naming/safe.go @@ -5,11 +5,23 @@ import ( "path/filepath" "strconv" "strings" + + "github.com/rclone/rclone/lib/encoder" ) -// maxNameBytes is NAME_MAX on Linux: the longest single path component ext4 and -// friends accept. It is a byte count, not a rune count. -const maxNameBytes = 255 +// maxNameBytes is the longest stored name that can actually be written. +// +// NAME_MAX on Linux is 255 bytes for a single path component, but the name is +// not what lands on disk first: a download is written to name+".part" and +// renamed afterwards, so the suffix has to fit inside the limit too. Checking +// the bare 255 would pass a name whose part file then fails with ENAMETOOLONG — +// after the item had already been queued, which stalls the walk on the same +// message on every pass. +const maxNameBytes = 255 - len(PartSuffix) + +// PartSuffix marks a download still in flight. It lives here because Safe's +// length limit has to account for it. +const PartSuffix = ".part" // Safe reports whether a stored name can be joined onto a directory path. // @@ -48,6 +60,26 @@ func Safe(name string) error { // Catches embedded separators, trailing slashes, and any ".." segment, // since Base of all of those differs from the original. return fmt.Errorf("filename is not a single path element: %q", name) + case encoder.OS.FromStandardName(name) != name: + // rclone does not address files by the bytes on disk. Every name given + // to an Fs is run through the backend's encoder, and every name listed + // back is re-encoded to the standard set — neither of which os.OpenFile + // performs. So a name containing one of the characters those encoders + // rewrite is written verbatim, then looked up under a different string: + // the upload fails with "object not found" forever, or it succeeds and + // the index records a name the presence check will never match. + // + // That is the original bug exactly — one name derived two ways — with + // rclone's encoder in the place filenamify used to occupy. Rejecting + // rather than encoding is the same choice made everywhere else here: + // encoding would give the two derivations a chance to disagree again. + return fmt.Errorf("filename is rewritten by rclone's path encoder: %q", name) + case encoder.Standard.Encode(encoder.Standard.Decode(name)) != name: + // The listing side of the same problem, and it is not backend-specific: + // the re-encode to the standard set happens above the backend encoder, + // so it applies to every remote. Control characters and DEL are the + // common case, and both are trivially settable in a Telegram filename. + return fmt.Errorf("filename is rewritten when rclone lists it back: %q", name) } return nil } diff --git a/internal/naming/safe_test.go b/internal/naming/safe_test.go index 531202d..766b873 100644 --- a/internal/naming/safe_test.go +++ b/internal/naming/safe_test.go @@ -1,6 +1,8 @@ package naming import ( + "fmt" + "os" "path/filepath" "strings" "testing" @@ -120,21 +122,30 @@ func TestSplitStoredDoesNotMatchPrefixOverlap(t *testing.T) { // An over-long name is the one input that can hang a drive-until-complete loop: // os.Create rejects it, so the download fails forever while verify keeps // reporting it absent. It must be refused up front, not discovered per pass. +// +// The limit leaves room for the ".part" suffix, because that is what is opened +// first. A name that fits in NAME_MAX but whose part file does not would pass +// this check, be queued, and then fail to open every single pass. func TestSafeRejectsNamesOverTheFilesystemLimit(t *testing.T) { prefix := "1234567890_42_" - fill := 255 - len(prefix) + fill := maxNameBytes - len(prefix) atLimit := prefix + strings.Repeat("a", fill) if err := Safe(atLimit); err != nil { t.Errorf("Safe(%d bytes) = %v, want nil at exactly the limit", len(atLimit), err) } + // The part file for a name at the limit must actually be creatable, which is + // the property the limit exists to guarantee. + if err := os.WriteFile(filepath.Join(t.TempDir(), atLimit+PartSuffix), nil, 0o600); err != nil { + t.Errorf("a name Safe accepted cannot be written as a part file: %v", err) + } overLimit := prefix + strings.Repeat("a", fill+1) err := Safe(overLimit) if err == nil { t.Fatalf("Safe(%d bytes) = nil, want an error past the limit", len(overLimit)) } - if !strings.Contains(err.Error(), "over the 255-byte limit") { + if !strings.Contains(err.Error(), fmt.Sprintf("over the %d-byte limit", maxNameBytes)) { t.Errorf("error should name the limit, got: %v", err) } @@ -145,3 +156,45 @@ func TestSafeRejectsNamesOverTheFilesystemLimit(t *testing.T) { len(multibyte), len([]rune(multibyte))) } } + +// rclone addresses files by an encoded name, not by the bytes os.OpenFile +// wrote. A name either encoder rewrites is the original two-derivations bug in +// a new place: written verbatim, then looked up or listed under a different +// string, so the file is re-fetched on every pass forever. +// +// These were confirmed against the real local backend before the check existed: +// NewObject on a written "a‛b.jpg" reported "object not found", and a written +// "a\nb.jpg" listed back as "a␊b.jpg". +func TestSafeRejectsNamesRcloneRewrites(t *testing.T) { + rejected := map[string]string{ + "the encoder's own escape character": "1234567890_42_a\u201bb.jpg", + "a symbol-for-control glyph": "1234567890_42_a\u2421b.jpg", + "a raw newline": "1234567890_42_a\nb.jpg", + "a raw DEL": "1234567890_42_a\x7fb.jpg", + "a raw control byte": "1234567890_42_a\x01b.jpg", + } + for label, name := range rejected { + t.Run(label, func(t *testing.T) { + if err := Safe(name); err == nil { + t.Errorf("Safe(%q) = nil; rclone rewrites this name", name) + } + }) + } + + // Rejecting too much would be its own bug: these are ordinary Telegram + // filenames and every one must survive. + accepted := []string{ + "1234567890_42_ünïcödé 12-08 🍓.mp4", + "1234567890_42_a#b%c!d[e]{f}.mp4", + "1234567890_42_ㅋㅋㅋ 😀.png", + "1234567890_42_a/b.jpg", // fullwidth solidus, not a separator + "1234567890_42_trailing. ", + "1234567890_42_'quoted' \"double\".mp4", + "1234567890_42_ünïcödé, spaces & commas.webm", + } + for _, name := range accepted { + if err := Safe(name); err != nil { + t.Errorf("Safe(%q) = %v, want nil for an ordinary filename", name, err) + } + } +} diff --git a/internal/pipeline/elem.go b/internal/pipeline/elem.go index 4738202..7e24c63 100644 --- a/internal/pipeline/elem.go +++ b/internal/pipeline/elem.go @@ -24,7 +24,11 @@ import ( // because upload is triggered by a download returning rather than by a filter // over a directory. A distinct suffix just keeps a staging directory shared with // a legacy tdl run unambiguous during the cutover. -const partSuffix = ".part" +// +// It is defined in naming because naming.Safe's length limit has to leave room +// for it — a name that fits but whose part file does not would pass the check +// and then fail to open, stalling the walk on that message forever. +const partSuffix = naming.PartSuffix // elem adapts one media item to the downloader's element interface. type elem struct { diff --git a/internal/remote/index.go b/internal/remote/index.go index 40acc9c..4e97e5c 100644 --- a/internal/remote/index.go +++ b/internal/remote/index.go @@ -54,14 +54,32 @@ type Index struct { // a narrowed listing here does not fail — it silently reports archived files as // absent and re-downloads every one of them. The transfer tunables in Init are // deliberately env-overridable; this is not. +// +// Note that filter.NewFilter(nil) does NOT give a neutral filter: it copies the +// package-level filter.Opt (filter.go:198-201), which rclone has already +// populated from RCLONE_* at init via RegisterGlobalOptions. Passing nil here +// reproduces exactly the inherited filter this is trying to escape, which is +// why every field that can narrow a listing is set explicitly. A zero-value +// Options is not a substitute either — it fails validation, because MinAge and +// MaxAge both being 0 reads as "min > max". func BuildIndex(ctx context.Context, f fs.Fs, dialogID int64) (*Index, error) { ctx, ci := fs.AddConfig(ctx) ci.MaxDepth = -1 - unfiltered, err := filter.NewFilter(nil) + unfiltered, err := filter.NewFilter(&filter.Options{ + MinAge: fs.DurationOff, + MaxAge: fs.DurationOff, + MinSize: -1, + MaxSize: -1, + }) if err != nil { return nil, fmt.Errorf("build an empty filter: %w", err) } + if !unfiltered.InActive() { + // Cheap and worth keeping: this is the assertion whose absence let a + // no-op neutralisation stand. + return nil, fmt.Errorf("internal: listing filter is not neutral") + } ctx = filter.ReplaceConfig(ctx, unfiltered) idx := &Index{ @@ -92,7 +110,13 @@ func BuildIndex(ctx context.Context, f fs.Fs, dialogID int64) (*Index, error) { // A destination that does not exist yet holds nothing. That is an empty // index, not a failure — it is what a first run against a new path looks // like, and treating it as an error would make verify unusable there. - if errors.Is(err, fs.ErrorDirNotFound) { + // + // Only when nothing was listed, though. rclone's walk records a failed + // directory and keeps going (walk.go:168-183), returning the error at the + // end, so this same error also means "one subdirectory could not be + // listed" — and swallowing that would return a partial index as + // authoritative, reporting everything under it absent. + if errors.Is(err, fs.ErrorDirNotFound) && len(idx.byName) == 0 { return idx, nil } return nil, fmt.Errorf("list %s: %w", f.String(), err) diff --git a/internal/remote/index_env_test.go b/internal/remote/index_env_test.go new file mode 100644 index 0000000..1fb4124 --- /dev/null +++ b/internal/remote/index_env_test.go @@ -0,0 +1,64 @@ +package remote + +import ( + "os" + "os/exec" + "path/filepath" + "testing" + + _ "github.com/rclone/rclone/backend/local" + "github.com/rclone/rclone/fs" +) + +// rclone reads RCLONE_* into its global config at package init, so an in-process +// t.Setenv is too late to prove anything. A subprocess is the only way to +// observe what BuildIndex actually does under an operator's environment — which +// is exactly why the original no-op neutralisation went unnoticed. +func TestBuildIndexIgnoresInheritedFilters(t *testing.T) { + if os.Getenv("GO_INDEX_ENV_CHILD") == "1" { + indexChild(t) + return + } + for _, env := range []string{ + "RCLONE_EXCLUDE=*.mp4", + "RCLONE_FILTER=- *.mp4", + "RCLONE_MIN_SIZE=1M", + "RCLONE_MAX_AGE=1h", + "RCLONE_MAX_DEPTH=1", + } { + t.Run(env, func(t *testing.T) { + cmd := exec.Command(os.Args[0], "-test.run=TestBuildIndexIgnoresInheritedFilters", "-test.v") + cmd.Env = append(os.Environ(), "GO_INDEX_ENV_CHILD=1", env) + out, err := cmd.CombinedOutput() + if err != nil { + t.Errorf("%s narrowed the index:\n%s", env, out) + } + }) + } +} + +func indexChild(t *testing.T) { + dir := t.TempDir() + for _, rel := range []string{"a.mp4", "b.txt", "sub/c.mp4"} { + p := filepath.Join(dir, rel) + if err := os.MkdirAll(filepath.Dir(p), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(p, []byte("x"), 0o600); err != nil { + t.Fatal(err) + } + } + f, err := fs.NewFs(t.Context(), dir) + if err != nil { + t.Fatal(err) + } + idx, err := BuildIndex(t.Context(), f, 1234567890) + if err != nil { + t.Fatalf("BuildIndex: %v", err) + } + for _, want := range []string{"a.mp4", "b.txt", "c.mp4"} { + if _, ok := idx.Lookup(want); !ok { + t.Errorf("%q missing from the index", want) + } + } +} diff --git a/internal/tgsource/chat.go b/internal/tgsource/chat.go index 7729c84..9e913c4 100644 --- a/internal/tgsource/chat.go +++ b/internal/tgsource/chat.go @@ -26,10 +26,10 @@ var telegramHosts = []string{"t.me/", "telegram.me/", "telegram.dog/"} // // This is parsed rather than pattern-matched because the two HTTPS shapes overlap // in a way a regex gets wrong: a private link is t.me/c// and a public -// one is t.me//, so "t.me/c/1234567890" — a perfectly good private -// *channel* link — looks exactly like a public message link with the username -// "c". The distinction is whether a trailing numeric component follows the chat, -// and where that component sits depends on the "c" marker. +// one is t.me//, so "t.me/c/1234567890" — a private channel link — +// looks exactly like a public message link with the username "c". The +// distinction is whether a trailing numeric component follows the chat, and +// where that component sits depends on the leading marker. func isMessageLink(s string) bool { lower := strings.ToLower(s) @@ -51,29 +51,82 @@ func isMessageLink(s string) bool { rest, _, _ = strings.Cut(rest, "#") parts := strings.Split(strings.Trim(rest, "/"), "/") - if len(parts) >= 1 && parts[0] == "c" { - // c/ is the channel; c// is one message in it. + if len(parts) >= 1 && (parts[0] == "c" || parts[0] == "s") { + // Both put the chat in the second component, so a message is the third: + // c// and s//. return len(parts) >= 3 && isDigits(parts[2]) } - // t.me/joinchat/ and t.me/s/ are chats, not messages, and their - // second component is not a bare number — except for a hypothetical all-digit - // invite hash, which is not worth mis-parsing every real link to guard. - if len(parts) >= 1 && (parts[0] == "joinchat" || parts[0] == "s") { + // t.me/joinchat/ is a chat, and its second component is not a bare + // number — except for a hypothetical all-digit invite hash, which is not + // worth mis-parsing every real link to guard. + if len(parts) >= 1 && parts[0] == "joinchat" { return false } return len(parts) >= 2 && isDigits(parts[1]) } -// afterHost returns the path following a Telegram host, if s names one. -func afterHost(lower string) (string, bool) { - for _, host := range telegramHosts { - if i := strings.Index(lower, host); i >= 0 { - return lower[i+len(host):], true +// linkChat extracts the chat from a link whose first path component is a marker +// rather than the chat itself. +// +// Without this the marker *is* the chat as far as the resolver is concerned. +// gotd's deeplink parser takes the first path component as the domain and drops +// the rest (deeplink.go:106-148), and ValidateDomain accepts a single letter, so +// "t.me/s/mychannel" resolves the username "s" — either a hard-to-read +// USERNAME_NOT_OCCUPIED, or, since one-character usernames exist, somebody +// else's chat archived into the operator's remote. +// +// t.me/s/ is the preview page for a public channel, and the form most +// likely to be copied out of a browser. t.me/c/ carries the bare MTProto +// channel id — the same value a -100 Bot API id strips to. +func linkChat(s string) (string, bool) { + lower := strings.ToLower(s) + if strings.HasPrefix(lower, "tg://") { + return "", false + } + i, ok := hostEnd(lower) + if !ok { + return "", false + } + rest := s[i:] + rest, _, _ = strings.Cut(rest, "?") + rest, _, _ = strings.Cut(rest, "#") + + parts := strings.Split(strings.Trim(rest, "/"), "/") + if len(parts) < 2 || parts[1] == "" { + return "", false + } + switch strings.ToLower(parts[0]) { + case "c": + if isDigits(parts[1]) { + return parts[1], true } + case "s": + return parts[1], true } return "", false } +// afterHost returns the path following a Telegram host, if s names one. +func afterHost(lower string) (string, bool) { + i, ok := hostEnd(lower) + if !ok { + return "", false + } + return lower[i:], true +} + +// hostEnd returns the offset just past a Telegram host in an already-lowercased +// string. Offsets rather than a substring, so a caller can slice the original +// and keep the chat's real case. +func hostEnd(lower string) (int, bool) { + for _, host := range telegramHosts { + if i := strings.Index(lower, host); i >= 0 { + return i + len(host), true + } + } + return 0, false +} + func isDigits(s string) bool { if s == "" { return false @@ -105,6 +158,10 @@ func NormalizeChat(chat string) (string, error) { "pass the chat's username or id instead", chat) } + if c, ok := linkChat(chat); ok { + return c, nil + } + if m := botAPIID.FindStringSubmatch(chat); m != nil { return m[1], nil } diff --git a/internal/tgsource/chat_test.go b/internal/tgsource/chat_test.go index d13deff..8354846 100644 --- a/internal/tgsource/chat_test.go +++ b/internal/tgsource/chat_test.go @@ -21,12 +21,17 @@ func TestNormalizeChat(t *testing.T) { {"tg protocol link", "tg://resolve?domain=mychannel", "tg://resolve?domain=mychannel"}, {"bot api id loses the -100 prefix", "-1001234567890", "1234567890"}, {"surrounding whitespace is trimmed", " mychannel\n", "mychannel"}, - {"private channel link without a message", "https://t.me/c/1234567890", "https://t.me/c/1234567890"}, - // Invite and preview links are chats; their second component is not a - // bare message number and must not be read as one. + // The "c" and "s" markers are not the chat. Passed through whole, gotd + // takes the first path component as the username and resolves "c" or + // "s" — a confusing failure at best, and somebody else's chat at worst, + // since one-character usernames exist. + {"private channel link yields the channel id", "https://t.me/c/1234567890", "1234567890"}, + {"preview link yields the username", "https://t.me/s/mychannel", "mychannel"}, + {"preview link keeps the username's case", "https://t.me/s/MyChannel", "MyChannel"}, + // Invite links are chats; their second component is not a bare message + // number and must not be read as one. {"invite link", "https://t.me/+AbCd_1234", "https://t.me/+AbCd_1234"}, {"joinchat link", "https://t.me/joinchat/AbCd1234", "https://t.me/joinchat/AbCd1234"}, - {"preview link", "https://t.me/s/mychannel", "https://t.me/s/mychannel"}, {"other telegram host, no message", "https://telegram.dog/mychannel", "https://telegram.dog/mychannel"}, {"tg link without a post parameter", "tg://resolve?domain=mychannel", "tg://resolve?domain=mychannel"}, } @@ -50,6 +55,7 @@ func TestNormalizeChatRejectsMessageLinks(t *testing.T) { for _, in := range []string{ "https://t.me/c/1234567890/4242", "t.me/c/1234567890/4242", + "https://t.me/s/mychannel/4242", "https://t.me/mychannel/4242", // gotd accepts all three Telegram hosts and its parser keeps only the // domain, silently dropping the message number — so missing one of these diff --git a/internal/tgsource/client.go b/internal/tgsource/client.go index 72de17d..72074d5 100644 --- a/internal/tgsource/client.go +++ b/internal/tgsource/client.go @@ -133,7 +133,7 @@ func (s *Session) Run(ctx context.Context, fn func(context.Context, dcpool.Pool) tclient.NewDefaultMiddlewares(ctx, s.timeout)...) defer func() { _ = pool.Close() }() - return fn(ctx, pool) + return fn(ctx, withSafeTakeout(pool)) }) // gotd swallows cancellation: telegram.Client.Run ends with diff --git a/internal/tgsource/takeout.go b/internal/tgsource/takeout.go new file mode 100644 index 0000000..5b763b2 --- /dev/null +++ b/internal/tgsource/takeout.go @@ -0,0 +1,62 @@ +package tgsource + +import ( + "context" + "sync" + + "github.com/gotd/td/tg" + + "github.com/iyear/tdl/core/dcpool" + "github.com/iyear/tdl/core/middlewares/takeout" +) + +// safeTakeout wraps a pool to make its takeout path survive a failed init. +// +// core's own dcpool.Takeout deadlocks on that path. It holds the pool's mutex +// for the whole call, and its recovery from a failed init is to return +// p.Client(ctx, dc) — which locks the same mutex again (dcpool.go:113-121 and +// :57-58). sync.Mutex is not reentrant, so the worker blocks forever, then every +// other worker blocks behind it, and the process hangs with no output and no +// response to cancellation, since the goroutine is parked on a mutex rather than +// a select. +// +// That is not an exotic path. Telegram answers account.initTakeoutSession with +// TAKEOUT_INIT_DELAY when a takeout was started recently — tdl's own "ignore +// init delay error" comment shows it expects exactly this — and takeout is on by +// default, so running two exports in succession is enough to trigger it. +// +// Probing before the run is not an alternative: a probe would consume an init +// and make the pool's own init the one that gets the delay error. So the takeout +// session is established here instead, once, and the pool's Takeout is never +// called at all. +type safeTakeout struct { + dcpool.Pool + + once sync.Once + id int64 + ok bool +} + +func withSafeTakeout(p dcpool.Pool) dcpool.Pool { return &safeTakeout{Pool: p} } + +// Takeout returns a takeout-scoped client, or an ordinary one if no takeout +// session could be established. +// +// Falling back rather than failing matches what core intended: takeout raises +// rate limits and reaches older history, but a download works without it. The +// difference is that this fallback returns. +func (s *safeTakeout) Takeout(ctx context.Context, dc int) *tg.Client { + base := s.Pool.Client(ctx, dc) + + s.once.Do(func() { + id, err := takeout.Takeout(ctx, base.Invoker()) + if err != nil { + return // ok stays false; every caller gets a plain client + } + s.id, s.ok = id, true + }) + if !s.ok { + return base + } + return tg.NewClient(takeout.Middleware(s.id).Handle(base.Invoker())) +} diff --git a/internal/tgsource/takeout_test.go b/internal/tgsource/takeout_test.go new file mode 100644 index 0000000..dd9274f --- /dev/null +++ b/internal/tgsource/takeout_test.go @@ -0,0 +1,64 @@ +package tgsource + +import ( + "context" + "errors" + "sync/atomic" + "testing" + "time" + + "github.com/gotd/td/bin" + "github.com/gotd/td/tg" + + "github.com/iyear/tdl/core/dcpool" +) + +// failingInvoker refuses every call, which is what Telegram does to +// account.initTakeoutSession with TAKEOUT_INIT_DELAY. +type failingInvoker struct{ calls atomic.Int64 } + +func (f *failingInvoker) Invoke(context.Context, bin.Encoder, bin.Decoder) error { + f.calls.Add(1) + return errors.New("TAKEOUT_INIT_DELAY_86400") +} + +type fakePool struct{ inv tg.Invoker } + +func (p *fakePool) Client(context.Context, int) *tg.Client { return tg.NewClient(p.inv) } +func (p *fakePool) Takeout(context.Context, int) *tg.Client { + panic("upstream Takeout must not be called") +} +func (p *fakePool) Default(context.Context) *tg.Client { return tg.NewClient(p.inv) } +func (p *fakePool) Close() error { return nil } + +// core's dcpool.Takeout holds the pool mutex and recovers from a failed init by +// calling Client, which locks the same mutex — so the upstream version of this +// test hangs instead of failing. The wrapper must return a usable client. +func TestSafeTakeoutFallsBackWhenInitFails(t *testing.T) { + inv := &failingInvoker{} + var pool dcpool.Pool = withSafeTakeout(&fakePool{inv: inv}) + + done := make(chan *tg.Client, 1) + go func() { done <- pool.Takeout(t.Context(), 2) }() + + select { + case got := <-done: + if got == nil { + t.Fatal("Takeout returned nil after a failed init") + } + case <-time.After(5 * time.Second): + t.Fatal("Takeout deadlocked after a failed init") + } + + // The init is attempted once, not once per file: a failing init that ran on + // every element would add a round trip to each of 18k downloads. + before := inv.calls.Load() + for range 5 { + if pool.Takeout(t.Context(), 2) == nil { + t.Fatal("Takeout returned nil") + } + } + if got := inv.calls.Load(); got != before { + t.Errorf("takeout init retried %d times after failing; want no retries", got-before) + } +} From aea29826c20c2a0f0ad6c393dfcc366e4322aea0 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Sun, 6 Sep 2026 20:44:08 +0700 Subject: [PATCH 08/16] fix: detect truncated uploads, and stop reporting unreachable work as retryable MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Verification matched on name and non-zero size, so an upload that died partway was counted archived permanently. Comparing against the size Telegram reports found six such objects in the live archive, one of them 221 MiB standing in for a 2006 MiB video. They are folded into the outstanding set; rclone overwrites a size mismatch, so another pass repairs them. A zero Report claimed the archive was complete — nothing expected, nothing missing — which is the value both commands hold before their Telegram callback populates it, so any early return printed COMPLETE and exited 0 on an untouched chat. A Report now knows whether it ran. A name that can never be written kept the run outstanding forever while the download set deliberately excluded it, so a driver looping on "incomplete" walked the whole history and re-indexed the whole remote every pass for work that could not be done. Such a run now reports STALLED and exits 4. A destination that stopped accepting uploads was reported and then discarded, exiting 1. The same driver would retry against a full or unreachable remote indefinitely, downloading gigabytes each pass to upload none. It exits 3. Free space is re-checked during the run, not only before it. An archive this size runs for hours, and the remote can fill in the middle; discovering it through five failed multi-gigabyte uploads wastes the download for all of them. Also: parseSize silently wrapped to a negative or zero on a large input, which reads downstream as "no cap"; list printed attacker-chosen filenames raw, so a tab shifted the columns and an escape sequence reached the terminal; a missing backend blamed credentials rather than the build; humanBytes indexed past its unit table above 1 PiB; a second Init reported success against a config that never loaded; and sync did not surface the basename collisions verify warned about, though sync is the command that acts on the verdict. The env-override test could not observe what it claimed: rclone reads RCLONE_* at package init, so t.Setenv came too late and the assertion held with the guard removed. It runs in a subprocess now, as does the new one covering the index against inherited filters. --- cmd/tgexport/list.go | 9 ++- cmd/tgexport/main.go | 9 +++ cmd/tgexport/main_test.go | 9 +++ cmd/tgexport/sync.go | 55 ++++++++++++++++- internal/pipeline/pipeline.go | 30 ++++++++-- internal/pipeline/run_test.go | 7 ++- internal/pipeline/space.go | 66 +++++++++++++++++++++ internal/pipeline/space_test.go | 76 ++++++++++++++++++++++++ internal/remote/fs.go | 54 +++++++++++++++-- internal/remote/fs_test.go | 65 ++++++++++++++++---- internal/report/progress.go | 8 ++- internal/tgsource/iterate.go | 10 ++++ internal/verify/verify.go | 102 ++++++++++++++++++++++++++++---- internal/verify/verify_test.go | 63 +++++++++++++++++++- 14 files changed, 519 insertions(+), 44 deletions(-) create mode 100644 internal/pipeline/space.go create mode 100644 internal/pipeline/space_test.go diff --git a/cmd/tgexport/list.go b/cmd/tgexport/list.go index 7afb020..81981af 100644 --- a/cmd/tgexport/list.go +++ b/cmd/tgexport/list.go @@ -19,8 +19,11 @@ import ( // // It is the smallest thing that exercises the whole read path — resolve a chat, // walk its history, derive a name — so a naming or paging problem shows up here -// rather than halfway through an archive run. The output is tab-separated on -// purpose: filenames contain spaces, commas and quotes, but not tabs. +// rather than halfway through an archive run. The output is tab-separated, and +// the name is quoted: it comes from DocumentAttributeFilename, so whoever +// uploaded the file chose it, and a raw tab would shift the columns while a raw +// newline would split the record. Quoting also renders escape sequences inert +// rather than letting them redraw the operator's terminal. func listCmd(ctx context.Context, args []string) error { fs := flag.NewFlagSet("list", flag.ContinueOnError) var ( @@ -71,7 +74,7 @@ func listCmd(ctx context.Context, args []string) error { } count++ total += it.Size() - if _, err := fmt.Fprintf(out, "%d\t%d\t%s\n", it.MessageID, it.Size(), it.Name); err != nil { + if _, err := fmt.Fprintf(out, "%d\t%d\t%q\n", it.MessageID, it.Size(), it.Name); err != nil { return err } } diff --git a/cmd/tgexport/main.go b/cmd/tgexport/main.go index 7314d8f..582e3b6 100644 --- a/cmd/tgexport/main.go +++ b/cmd/tgexport/main.go @@ -30,6 +30,7 @@ const ( exitIncomplete = 1 exitUsage = 2 exitRemoteError = 3 + exitStalled = 4 exitSIGINT = 130 exitSIGTERM = 143 ) @@ -41,6 +42,12 @@ var errUsage = errors.New("usage") // errIncomplete marks a run that finished cleanly but left work outstanding. var errIncomplete = errors.New("incomplete") +// errStalled marks a run with work outstanding that no future run can do — every +// remaining file has a name that cannot be written. It is distinct from +// errIncomplete because a driver retrying on "incomplete" would otherwise walk +// the whole history and re-index the whole remote forever, achieving nothing. +var errStalled = errors.New("stalled") + func main() { os.Exit(run()) } @@ -102,6 +109,8 @@ func exitCode(err error, sig os.Signal) int { return exitSIGINT case errors.Is(err, errUsage): return exitUsage + case errors.Is(err, errStalled): + return exitStalled case errors.Is(err, errIncomplete): return exitIncomplete default: diff --git a/cmd/tgexport/main_test.go b/cmd/tgexport/main_test.go index dfd0de3..3aa2ad2 100644 --- a/cmd/tgexport/main_test.go +++ b/cmd/tgexport/main_test.go @@ -8,6 +8,8 @@ import ( "os" "syscall" "testing" + + "github.com/tiennm99dev/telegram-exporter/internal/pipeline" "time" ) @@ -27,6 +29,13 @@ func TestExitCode(t *testing.T) { {"wrapped help", fmt.Errorf("parse: %w", flag.ErrHelp), nil, exitOK}, {"usage mistake", fmt.Errorf("%w: bad flag", errUsage), nil, exitUsage}, {"run left work outstanding", fmt.Errorf("%w: 12 files", errIncomplete), nil, exitIncomplete}, + // Distinct from incomplete on purpose: a driver retrying on 1 would + // walk the whole chat forever for work that can never be done. + {"nothing left that can be fetched", fmt.Errorf("%w: 2 files", errStalled), nil, exitStalled}, + // And a destination that stopped accepting uploads is a failure, not a + // retry: the next pass would be refused identically. + {"destination refusing uploads", + fmt.Errorf("run: %w", pipeline.ErrDestinationFailing), nil, exitRemoteError}, {"remote failure", errors.New("pikpak unreachable"), nil, exitRemoteError}, {"cancelled without a signal", context.Canceled, nil, exitSIGINT}, {"wrapped cancellation", fmt.Errorf("download: %w", context.Canceled), nil, exitSIGINT}, diff --git a/cmd/tgexport/sync.go b/cmd/tgexport/sync.go index d28b726..b87f6c1 100644 --- a/cmd/tgexport/sync.go +++ b/cmd/tgexport/sync.go @@ -6,6 +6,7 @@ import ( "flag" "fmt" "iter" + "math" "os" "github.com/iyear/tdl/core/dcpool" @@ -108,7 +109,10 @@ func syncCmd(ctx context.Context, args []string) error { return err } - var final verify.Report + var ( + final verify.Report + runErr error + ) if err := sess.Run(ctx, func(ctx context.Context, pool dcpool.Pool) error { api := pool.Default(ctx) @@ -130,6 +134,7 @@ func syncCmd(ctx context.Context, args []string) error { if err != nil { return err } + warnCollisions(idx) before := verify.Check(items, idx) fmt.Fprintf(os.Stderr, "%d media messages, %d already archived, %d to fetch\n", before.Expected, before.Present, len(before.Todo())) @@ -151,7 +156,8 @@ func syncCmd(ctx context.Context, args []string) error { fmt.Fprintf(os.Stderr, "fetching %d file(s), %.1f GiB\n", len(todo), float64(todoBytes)/(1<<30)) rep := report.New(os.Stderr, len(todo), todoBytes) - res, runErr := pipeline.Run(ctx, sliceSeq(todo), pipeline.Options{ + var res pipeline.Result + res, runErr = pipeline.Run(ctx, sliceSeq(todo), pipeline.Options{ Pool: pool, Dst: dst, Staging: *staging, @@ -162,6 +168,12 @@ func syncCmd(ctx context.Context, args []string) error { Confirm: *confirm, Takeout: *takeout, Report: rep.Update, + // Re-checked during the run, not only before it: an archive of this + // size runs for hours, and the destination can fill in the middle. + FreeBytes: func(ctx context.Context) (int64, bool) { + return remote.FreeBytes(ctx, dst) + }, + MinFree: *minFree * (1 << 30), }) rep.Finish(res.Stats) @@ -173,6 +185,7 @@ func syncCmd(ctx context.Context, args []string) error { // and hit one transient upload error has still made progress, and // suppressing the report would leave the operator — and any driver // reading the exit code — unable to tell that from a total failure. + // It is returned after the report, below, so the exit code is right. fmt.Fprintf(os.Stderr, "run ended early: %v\n", runErr) } @@ -191,7 +204,17 @@ func syncCmd(ctx context.Context, args []string) error { fmt.Fprintln(os.Stderr) final.Write(os.Stdout) - if !final.Complete() { + switch { + case errors.Is(runErr, pipeline.ErrDestinationFailing): + // Exit 3, not 1. The destination refused upload after upload, and it + // will refuse them next pass too — a driver retrying on "incomplete" + // would walk 18k messages and re-download gigabytes into a remote that + // cannot take a byte, indefinitely. + return runErr + case final.Stalled(): + return fmt.Errorf("%w: %d file(s) remain, none of which can be fetched", + errStalled, len(final.Todo())) + case !final.Complete(): return fmt.Errorf("%w: %d file(s) still to fetch", errIncomplete, len(final.Todo())) } return nil @@ -257,6 +280,23 @@ func validateBudget(budget int64, todo []tgsource.Item) error { return nil } +// warnCollisions reports basenames the index found at more than one path. +// +// verify printed this and sync did not, which was backwards: an ambiguous +// snapshot makes the presence verdict for those names unreliable, and sync is +// the command that acts on the verdict by moving data. +func warnCollisions(idx *remote.Index) { + dup := idx.Collisions() + if len(dup) == 0 { + return + } + fmt.Fprintf(os.Stderr, "warning: %d basename(s) appear at more than one path; "+ + "verdicts for them may flip between runs:\n", len(dup)) + for _, name := range dup[:min(5, len(dup))] { + fmt.Fprintf(os.Stderr, " %q\n", name) + } +} + // checkFree refuses to start when the remote is nearly full. // // A backend that cannot report a quota is treated as unlimited rather than as a @@ -317,10 +357,19 @@ func parseSize(s string) (int64, error) { if r < '0' || r > '9' { return 0, fmt.Errorf("%q is not a size", s) } + // Checked rather than allowed to wrap. A wrapped value is negative or + // zero, and both mean "no cap" downstream — so a typo would silently + // remove the staging limit instead of being refused. + if n > (math.MaxInt64-int64(r-'0'))/10 { + return 0, fmt.Errorf("%q is too large", s) + } n = n*10 + int64(r-'0') } if n <= 0 { return 0, fmt.Errorf("must be greater than zero") } + if n > math.MaxInt64/mult { + return 0, fmt.Errorf("%q is too large", s) + } return n * mult, nil } diff --git a/internal/pipeline/pipeline.go b/internal/pipeline/pipeline.go index 63a2487..a0d6cc7 100644 --- a/internal/pipeline/pipeline.go +++ b/internal/pipeline/pipeline.go @@ -36,6 +36,12 @@ type Options struct { // Zero uses the shell pipeline's default of 5. MaxFailures int + // FreeBytes and MinFree stop the run when the destination fills mid-way. + // Both must be set for the check to happen; a backend that cannot report a + // quota counts as unlimited. + FreeBytes func(context.Context) (int64, bool) + MinFree int64 + Report func(Stats) } @@ -56,6 +62,14 @@ func (r Result) Failed() []Outcome { return out } +// ErrDestinationFailing marks a run stopped because the destination refused +// upload after upload. It is distinct from an ordinary upload failure because +// the right response differs: a transient error is worth retrying, while a +// remote that is full, unreachable, or refusing credentials will refuse the next +// pass identically, and a driver that retries walks the whole chat and downloads +// gigabytes for nothing every time. +var ErrDestinationFailing = errors.New("destination stopped accepting uploads") + // maxRecordedErrors bounds what a run keeps from a failing remote. Past this, // the pattern is established and joining thousands of identical strings just // makes the final message unreadable. @@ -102,6 +116,7 @@ func Run(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Options) (R up := &uploader{local: local, dst: o.Dst, confirm: o.Confirm} budget := newBudget(o.Budget) + guard := newSpaceGuard(o.FreeBytes, o.MinFree) uploads := make(chan tgsource.Item, o.Uploads) var ( @@ -126,7 +141,12 @@ func Run(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Options) (R go func() { defer wg.Done() for it := range uploads { - err := up.upload(ctx, it) + // A full destination is not worth another multi-gigabyte + // attempt, so the file is dropped rather than uploaded. + err := guard.check(ctx) + if err == nil { + err = up.upload(ctx, it) + } if err != nil { // MoveFile leaves the local copy in place when it fails, so @@ -146,7 +166,9 @@ func Run(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Options) (R } nErrs++ streak++ - if streak >= o.MaxFailures && !tripped { + // A full remote trips immediately: unlike a transient + // error, waiting for a streak just wastes the downloads. + if (streak >= o.MaxFailures || errors.Is(err, ErrDestinationFailing)) && !tripped { tripped = true stopDownloads.Store(true) } @@ -195,8 +217,8 @@ func Run(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Options) (R var upErr error switch { case trip: - upErr = fmt.Errorf("stopped after %d consecutive upload failures (%d total): %w", - o.MaxFailures, total, joined) + upErr = fmt.Errorf("%w: stopped after %d upload failure(s): %w", + ErrDestinationFailing, total, joined) case total > 0: upErr = fmt.Errorf("%d upload(s) failed: %w", total, joined) } diff --git a/internal/pipeline/run_test.go b/internal/pipeline/run_test.go index e0841e3..44f9f7b 100644 --- a/internal/pipeline/run_test.go +++ b/internal/pipeline/run_test.go @@ -7,7 +7,6 @@ import ( "iter" "os" "path/filepath" - "strings" "sync/atomic" "testing" @@ -239,8 +238,10 @@ func TestRunStopsDownloadingAfterConsecutiveUploadFailures(t *testing.T) { if err == nil { t.Fatal("Run returned nil, want the breaker's error") } - if !strings.Contains(err.Error(), "consecutive upload failures") { - t.Errorf("error does not mention the breaker: %v", err) + // The sentinel, not the wording: the caller maps this to a distinct exit + // code so a driver stops instead of retrying against a dead remote. + if !errors.Is(err, ErrDestinationFailing) { + t.Errorf("error is not ErrDestinationFailing: %v", err) } // The point of stopping the iterator rather than cancelling uploads: the // download side must not have walked the whole chat. diff --git a/internal/pipeline/space.go b/internal/pipeline/space.go new file mode 100644 index 0000000..c2125d1 --- /dev/null +++ b/internal/pipeline/space.go @@ -0,0 +1,66 @@ +package pipeline + +import ( + "context" + "fmt" + "sync" + "time" +) + +// spaceCheckInterval is how often the destination's free space is re-read +// mid-run. About is a network round trip, so it is not worth doing per file. +const spaceCheckInterval = time.Minute + +// spaceGuard watches the destination's free space during a run. +// +// Checking only before starting is not enough on a long archive: an 18k-message +// chat runs for hours, and a remote that was fine at the start can fill in the +// middle — from this run's own uploads, or from anything else using the account. +// Without this the run discovers it by failing several multi-gigabyte uploads in +// a row, which costs the download bandwidth for all of them. +// +// A backend that cannot report a quota is treated as unlimited, matching the +// pre-flight check and the shell pipeline before it. +type spaceGuard struct { + free func(context.Context) (int64, bool) + minFree int64 + + mu sync.Mutex + lastCheck time.Time + failed error +} + +func newSpaceGuard(free func(context.Context) (int64, bool), minFree int64) *spaceGuard { + if free == nil || minFree <= 0 { + return nil + } + return &spaceGuard{free: free, minFree: minFree, lastCheck: time.Now()} +} + +// check reports an error once the destination has dropped below the floor. +// +// The verdict is sticky: once the remote is known to be full, every later call +// says so without another round trip, because the run is ending either way. +func (g *spaceGuard) check(ctx context.Context) error { + if g == nil { + return nil + } + g.mu.Lock() + defer g.mu.Unlock() + + if g.failed != nil { + return g.failed + } + if time.Since(g.lastCheck) < spaceCheckInterval { + return nil + } + g.lastCheck = time.Now() + + free, ok := g.free(ctx) + if !ok || free >= g.minFree { + return nil + } + g.failed = fmt.Errorf("%w: only %.1f GiB free, below the %.1f GiB floor", + ErrDestinationFailing, float64(free)/(1<<30), float64(g.minFree)/(1<<30)) + return g.failed +} diff --git a/internal/pipeline/space_test.go b/internal/pipeline/space_test.go new file mode 100644 index 0000000..87c4057 --- /dev/null +++ b/internal/pipeline/space_test.go @@ -0,0 +1,76 @@ +package pipeline + +import ( + "context" + "errors" + "testing" + "time" +) + +func TestSpaceGuard(t *testing.T) { + const floor = 10 << 30 + + t.Run("unset guard never complains", func(t *testing.T) { + if err := newSpaceGuard(nil, floor).check(t.Context()); err != nil { + t.Errorf("check = %v, want nil when no reporter is configured", err) + } + }) + + t.Run("a backend without a quota counts as unlimited", func(t *testing.T) { + g := newSpaceGuard(func(context.Context) (int64, bool) { return 0, false }, floor) + g.lastCheck = time.Now().Add(-2 * spaceCheckInterval) + if err := g.check(t.Context()); err != nil { + t.Errorf("check = %v, want nil when the backend cannot report", err) + } + }) + + t.Run("plenty of room is fine", func(t *testing.T) { + g := newSpaceGuard(func(context.Context) (int64, bool) { return 100 << 30, true }, floor) + g.lastCheck = time.Now().Add(-2 * spaceCheckInterval) + if err := g.check(t.Context()); err != nil { + t.Errorf("check = %v, want nil with 100 GiB free", err) + } + }) + + t.Run("below the floor stops the run", func(t *testing.T) { + var calls int + g := newSpaceGuard(func(context.Context) (int64, bool) { + calls++ + return 1 << 30, true + }, floor) + g.lastCheck = time.Now().Add(-2 * spaceCheckInterval) + + err := g.check(t.Context()) + if !errors.Is(err, ErrDestinationFailing) { + t.Fatalf("check = %v, want ErrDestinationFailing", err) + } + // Sticky, and without another round trip: the run is ending either way, + // and every upload worker calls this. + for range 5 { + if !errors.Is(g.check(t.Context()), ErrDestinationFailing) { + t.Fatal("the verdict did not stick") + } + } + if calls != 1 { + t.Errorf("free space was read %d times, want 1", calls) + } + }) + + t.Run("checks are throttled", func(t *testing.T) { + var calls int + g := newSpaceGuard(func(context.Context) (int64, bool) { + calls++ + return 100 << 30, true + }, floor) + // Freshly built, so the interval has not elapsed. About is a network + // round trip and every worker calls this per file. + for range 20 { + if err := g.check(t.Context()); err != nil { + t.Fatalf("check = %v", err) + } + } + if calls != 0 { + t.Errorf("free space was read %d times inside the interval, want 0", calls) + } + }) +} diff --git a/internal/remote/fs.go b/internal/remote/fs.go index 1ab57b0..f84861a 100644 --- a/internal/remote/fs.go +++ b/internal/remote/fs.go @@ -6,6 +6,7 @@ import ( "errors" "fmt" "os" + "sort" "strings" "sync" @@ -33,7 +34,13 @@ func DefaultTunables() Tunables { // installOnce guards configfile.Install, which swaps unsynchronised package // globals in rclone's config package. Calling it twice is harmless on its own, // but racing it against an Fs resolution is not. -var installOnce sync.Once +var ( + installOnce sync.Once + // installErr is remembered, not just returned once. A second Init would + // otherwise skip the Do body and report success against a config that was + // never loaded. + installErr error +) // Init loads the user's rclone.conf and applies tunables to a derived context. // @@ -48,16 +55,21 @@ var installOnce sync.Once // code this tool defines as "incomplete", which would send a driver into an // endless retry. Loading up front turns that into an ordinary error. func Init(ctx context.Context, t Tunables) (context.Context, error) { - var err error installOnce.Do(func() { configfile.Install() if lerr := config.Data().Load(); lerr != nil && !errors.Is(lerr, config.ErrorConfigFileNotFound) { - err = fmt.Errorf("cannot read rclone config %q: %w "+ + installErr = fmt.Errorf("cannot read rclone config %q: %w "+ "(an encrypted config needs RCLONE_CONFIG_PASS)", config.GetConfigPath(), lerr) + return } + // Marks the data loaded so rclone's own lazy path is never taken. That + // path ends in fs.Fatalf -> os.Exit(1), past every defer and with an + // exit code this tool defines as "incomplete"; loading here and not + // flagging it would leave the window open until the first NewFs. + config.LoadedData() }) - if err != nil { - return ctx, err + if installErr != nil { + return ctx, installErr } ctx, ci := fs.AddConfig(ctx) @@ -93,11 +105,43 @@ func Resolve(ctx context.Context, remote string) (fs.Fs, error) { return nil, fmt.Errorf("rclone remote %q is not configured; configured remotes: %s", remote, strings.Join(sections(), ", ")) } + if missing := missingBackend(err); missing != "" { + // Otherwise this reads as a credentials problem, which sends the + // operator to rclone config for a fault that is in the build. + return nil, fmt.Errorf("remote %q needs the %q backend, which this binary does "+ + "not contain — it registers only %s. A build without -tags slim includes "+ + "every rclone backend: %w", remote, missing, strings.Join(backends(), ", "), err) + } return nil, fmt.Errorf("cannot reach %q — check credentials and connectivity: %w", remote, err) } return f, nil } +// missingBackend names the backend rclone could not find, if that is what went +// wrong. fs.Find returns a bare fmt.Errorf with no sentinel to match, so the +// text is all there is to go on. +func missingBackend(err error) string { + const prefix = "didn't find backend called " + msg := err.Error() + i := strings.Index(msg, prefix) + if i < 0 { + return "" + } + name := strings.TrimPrefix(msg[i+len(prefix):], "\"") + name, _, _ = strings.Cut(name, "\"") + return name +} + +// backends lists the backends compiled into this binary. +func backends() []string { + out := make([]string, 0, len(fs.Registry)) + for _, info := range fs.Registry { + out = append(out, info.Name) + } + sort.Strings(out) + return out +} + // EnsureDir creates the destination if it is not there yet. // // A destination that does not exist yet is the normal case for a first run, and diff --git a/internal/remote/fs_test.go b/internal/remote/fs_test.go index 3dbd30c..310f712 100644 --- a/internal/remote/fs_test.go +++ b/internal/remote/fs_test.go @@ -3,6 +3,8 @@ package remote import ( "context" "os" + "os/exec" + "slices" "strings" "testing" @@ -92,18 +94,61 @@ func TestInitAppliesTunables(t *testing.T) { } } -// An operator's environment override must survive Init, so a tunable is only -// applied when the corresponding variable is absent. +// rclone reads RCLONE_TRANSFERS into its global config at package init, so an +// in-process t.Setenv cannot observe whether Init honours it — the assertion +// would pass with the guard removed. A subprocess is the only real check. func TestInitLeavesEnvOverridesAlone(t *testing.T) { - unset(t, "RCLONE_TRANSFERS") - t.Setenv("RCLONE_TRANSFERS", "7") - - ctx, err := Init(context.Background(), Tunables{Transfers: 2, LowLevelRetries: 20}) - if err != nil { - t.Fatalf("Init: %v", err) + if os.Getenv("GO_INIT_ENV_CHILD") == "1" { + ctx, err := Init(t.Context(), Tunables{Transfers: 2, LowLevelRetries: 20}) + if err != nil { + t.Fatalf("Init: %v", err) + } + ci := fs.GetConfig(ctx) + if ci.Transfers != 7 { + t.Errorf("Transfers = %d, want the operator's 7", ci.Transfers) + } + if ci.LowLevelRetries != 99 { + t.Errorf("LowLevelRetries = %d, want the operator's 99", ci.LowLevelRetries) + } + return } - if got := fs.GetConfig(ctx).Transfers; got == 2 { - t.Errorf("Transfers = 2; Init overwrote the RCLONE_TRANSFERS override") + + cmd := exec.Command(os.Args[0], "-test.run=TestInitLeavesEnvOverridesAlone", "-test.v") + cmd.Env = append(os.Environ(), + "GO_INIT_ENV_CHILD=1", + "RCLONE_TRANSFERS=7", + "RCLONE_LOW_LEVEL_RETRIES=99", + ) + if out, err := cmd.CombinedOutput(); err != nil { + t.Errorf("Init overrode the operator's environment:\n%s", out) + } +} + +// And with nothing set, the pikpak-derived tunables are what apply. +func TestInitAppliesTunablesWhenTheEnvIsQuiet(t *testing.T) { + if os.Getenv("GO_INIT_QUIET_CHILD") == "1" { + ctx, err := Init(t.Context(), Tunables{Transfers: 2, LowLevelRetries: 20}) + if err != nil { + t.Fatalf("Init: %v", err) + } + ci := fs.GetConfig(ctx) + if ci.Transfers != 2 || ci.LowLevelRetries != 20 { + t.Errorf("Transfers=%d LowLevelRetries=%d, want 2 and 20", + ci.Transfers, ci.LowLevelRetries) + } + return + } + + cmd := exec.Command(os.Args[0], "-test.run=TestInitAppliesTunablesWhenTheEnvIsQuiet", "-test.v") + cmd.Env = append(os.Environ(), "GO_INIT_QUIET_CHILD=1") + // Cleared rather than assumed absent: the parent's own environment may + // carry them, which would make this assert the opposite of what it says. + cmd.Env = slices.DeleteFunc(cmd.Env, func(kv string) bool { + return strings.HasPrefix(kv, "RCLONE_TRANSFERS=") || + strings.HasPrefix(kv, "RCLONE_LOW_LEVEL_RETRIES=") + }) + if out, err := cmd.CombinedOutput(); err != nil { + t.Errorf("Init did not apply its tunables:\n%s", out) } } diff --git a/internal/report/progress.go b/internal/report/progress.go index 69cf2c7..4d31ec9 100644 --- a/internal/report/progress.go +++ b/internal/report/progress.go @@ -93,12 +93,16 @@ func humanBytes(n int64) string { if n < unit { return fmt.Sprintf("%d B", n) } + // The unit table runs to exabytes so the index cannot escape it. A PiB is + // not reachable from a Telegram chat, but a panic in the progress line would + // take down a run that was working. + const units = "KMGTPE" div, exp := int64(unit), 0 - for v := n / unit; v >= unit; v /= unit { + for v := n / unit; v >= unit && exp < len(units)-1; v /= unit { div *= unit exp++ } - return fmt.Sprintf("%.1f %ciB", float64(n)/float64(div), "KMGT"[exp]) + return fmt.Sprintf("%.1f %ciB", float64(n)/float64(div), units[exp]) } // isTerminal reports whether w is a character device. diff --git a/internal/tgsource/iterate.go b/internal/tgsource/iterate.go index 53d2d66..87c90db 100644 --- a/internal/tgsource/iterate.go +++ b/internal/tgsource/iterate.go @@ -76,6 +76,16 @@ func Walk(ctx context.Context, api *tg.Client, peer peers.Peer) iter.Seq2[Item, if err := it.Err(); err != nil { yield(Item{}, fmt.Errorf("walk chat history: %w", err)) } + + // There is no cross-check that the walk saw the whole history, and the + // obvious one does not work. gotd's iterator ends with a nil error if a + // fetch yields an empty buffer (messages/iter.go:97,103,155-160), so a + // truncated walk is indistinguishable from a complete one — but + // Iterator.Total is the server's history count, which includes the + // deleted slots that Next skips at iter.go:159-162. Comparing the two + // reports a short read on any chat that has ever had a message deleted, + // which is nearly all of them. A real check would need a count of + // non-empty messages, which the API does not offer. } } diff --git a/internal/verify/verify.go b/internal/verify/verify.go index c72c72e..cac0b29 100644 --- a/internal/verify/verify.go +++ b/internal/verify/verify.go @@ -63,6 +63,13 @@ type Tiny struct { Size int64 } +// Mismatch is a present file whose stored size is not the size Telegram reports. +type Mismatch struct { + MessageID int + Name string + Want, Got int64 +} + // Report is the outcome of comparing a chat against a remote. type Report struct { Expected int // media messages in the chat @@ -72,25 +79,71 @@ type Report struct { Absent []int // not on the remote under the wanted name ZeroByte []int // present but empty - Misnamed []Misnamed - Unsafe []Unsafe - Tiny []Tiny + Misnamed []Misnamed + Unsafe []Unsafe + Tiny []Tiny + Mismatched []Mismatch + + // checked records that Check actually ran. Without it a zero Report claims + // the archive is complete — nothing expected, nothing missing — which is the + // value a command holds before its Telegram callback has populated it. Any + // path that returns early therefore reports success on an untouched chat. + checked bool } +// Ran reports whether this came from a Check rather than being a zero value. +func (r Report) Ran() bool { return r.checked } + // Todo lists the message ids needing another fetch, in ascending order. // -// Zero-byte files are included: rclone overwrites a size-mismatched destination, -// so simply fetching again repairs them. +// Zero-byte and wrong-size files are included: rclone overwrites a +// size-mismatched destination, so simply fetching again repairs them. func (r Report) Todo() []int { - todo := make([]int, 0, len(r.Absent)+len(r.ZeroByte)) + todo := make([]int, 0, len(r.Absent)+len(r.ZeroByte)+len(r.Mismatched)) todo = append(todo, r.Absent...) todo = append(todo, r.ZeroByte...) + for _, m := range r.Mismatched { + todo = append(todo, m.MessageID) + } sort.Ints(todo) return todo } +// Fetchable lists the outstanding ids a run could actually retrieve. +// +// It is Todo minus the unsafe names, which are in Todo because they are not +// archived and out of this because no run will ever archive them. The gap +// between the two is what tells "keep going" apart from "this is as far as it +// goes". +func (r Report) Fetchable() []int { + if len(r.Unsafe) == 0 { + return r.Todo() + } + blocked := make(map[int]struct{}, len(r.Unsafe)) + for _, u := range r.Unsafe { + blocked[u.MessageID] = struct{}{} + } + todo := r.Todo() + out := todo[:0:0] + for _, id := range todo { + if _, skip := blocked[id]; !skip { + out = append(out, id) + } + } + return out +} + // Complete reports whether every expected file is present and non-empty. -func (r Report) Complete() bool { return len(r.Todo()) == 0 } +func (r Report) Complete() bool { return r.checked && len(r.Todo()) == 0 } + +// Stalled reports that work remains and none of it can ever be done. +// +// This is the state a drive-until-complete loop cannot detect for itself: the +// report is identical on every pass, so a driver retrying on "incomplete" walks +// the whole history and indexes the whole remote forever, achieving nothing. +func (r Report) Stalled() bool { + return r.checked && len(r.Todo()) > 0 && len(r.Fetchable()) == 0 +} // Check compares the wanted items against an index of the remote. // @@ -99,7 +152,7 @@ func (r Report) Complete() bool { return len(r.Todo()) == 0 } // and the near-misses are collected into Misnamed rather than being quietly // accepted, because the re-download lands beside them and both copies stay. func Check(items []tgsource.Item, idx Index) Report { - r := Report{Expected: len(items)} + r := Report{Expected: len(items), checked: true} for _, it := range items { r.Bytes += it.Size() @@ -123,6 +176,16 @@ func Check(items []tgsource.Item, idx Index) Report { } case size == 0: r.ZeroByte = append(r.ZeroByte, it.MessageID) + case size != it.Size(): + // The remaining way a report could say "complete" when it is not. + // An upload that died partway leaves a plausible object under the + // right name, and matching on name and non-zero size alone would + // count it archived permanently. Telegram's size is known here, so + // there is no reason not to use it; rclone overwrites a mismatched + // destination, so fetching again repairs it. + r.Mismatched = append(r.Mismatched, Mismatch{ + MessageID: it.MessageID, Name: it.Name, Want: it.Size(), Got: size, + }) default: r.Present++ if size < tinyThreshold { @@ -140,6 +203,13 @@ func (r Report) Write(w io.Writer) { fmt.Fprintf(w, "present and intact : %d\n", r.Present) fmt.Fprintf(w, " absent : %d\n", len(r.Absent)) fmt.Fprintf(w, " zero-byte : %d\n", len(r.ZeroByte)) + if len(r.Mismatched) > 0 { + fmt.Fprintf(w, " wrong size : %d\n", len(r.Mismatched)) + for _, m := range r.Mismatched[:min(5, len(r.Mismatched))] { + fmt.Fprintf(w, " id %d %d B on the remote, expected %d %q\n", + m.MessageID, m.Got, m.Want, m.Name) + } + } if len(r.Tiny) > 0 { fmt.Fprintf(w, " under 1KiB (check, not retried): %d\n", len(r.Tiny)) @@ -149,8 +219,8 @@ func (r Report) Write(w io.Writer) { } if len(r.Unsafe) > 0 { - fmt.Fprintf(w, "\nunsafe filenames : %d\n", len(r.Unsafe)) - fmt.Fprintf(w, " these cannot be written to a path and are never fetched:\n") + fmt.Fprintf(w, "\nunarchivable : %d\n", len(r.Unsafe)) + fmt.Fprintf(w, " these names cannot be written and will never be fetched:\n") for _, u := range r.Unsafe { fmt.Fprintf(w, " id %d %v\n", u.MessageID, u.Reason) } @@ -168,9 +238,15 @@ func (r Report) Write(w io.Writer) { } } - if todo := r.Todo(); len(todo) > 0 { + todo := r.Todo() + switch { + case r.Stalled(): + fmt.Fprintf(w, "\nSTALLED: %d file(s) remain, none of which can be fetched.\n", len(todo)) + case len(todo) > 0: fmt.Fprintf(w, "\nneeds another pass : %d (ids %d–%d)\n", len(todo), todo[0], todo[len(todo)-1]) - return + case !r.checked: + fmt.Fprintf(w, "\nno chat was checked.\n") + default: + fmt.Fprintf(w, "\nCOMPLETE: every media message is present and non-empty.\n") } - fmt.Fprintf(w, "\nCOMPLETE: every media message is present and non-empty.\n") } diff --git a/internal/verify/verify_test.go b/internal/verify/verify_test.go index 6978b77..7196a61 100644 --- a/internal/verify/verify_test.go +++ b/internal/verify/verify_test.go @@ -157,9 +157,17 @@ func TestCheckFlagsUnsafeNames(t *testing.T) { var sb strings.Builder r.Write(&sb) - if !strings.Contains(sb.String(), "unsafe filenames") { + if !strings.Contains(sb.String(), "unarchivable") { t.Errorf("report should flag the unsafe name, got:\n%s", sb.String()) } + // Nothing else is outstanding, so this run is as far as it can get. Saying + // "needs another pass" here is what makes a driver loop forever. + if !r.Stalled() { + t.Error("a report whose only outstanding item is unarchivable must be stalled") + } + if !strings.Contains(sb.String(), "STALLED") { + t.Errorf("report should say it is stalled, got:\n%s", sb.String()) + } } // An unsafe name must be rejected even when something is stored under that @@ -179,3 +187,56 @@ func TestReportTodoIsSorted(t *testing.T) { t.Errorf("Todo() = %v, want %v", r.Todo(), want) } } + +// A zero Report is the value a command holds before its Telegram callback has +// run. It must not claim the archive is complete. +func TestZeroReportIsNotComplete(t *testing.T) { + var r Report + if r.Complete() { + t.Error("a Report that never ran reports the archive complete") + } + if r.Ran() { + t.Error("Ran() is true on a zero Report") + } + if r.Stalled() { + t.Error("a Report that never ran reports itself stalled") + } + + var sb strings.Builder + r.Write(&sb) + if strings.Contains(sb.String(), "COMPLETE") { + t.Errorf("a Report that never ran printed COMPLETE:\n%s", sb.String()) + } +} + +// A remote object under the right name but the wrong size is the last way a +// report could say complete when it is not: an upload that died partway leaves +// exactly that, and the local copy is already gone. +func TestCheckTreatsAWrongSizeAsOutstanding(t *testing.T) { + items := []tgsource.Item{item(4242, "clip.mp4", 4096)} + idx := fakeIndex{"1234567890_4242_clip.mp4": 400} + + r := Check(items, idx) + if r.Present != 0 { + t.Errorf("Present = %d, want 0 for a truncated object", r.Present) + } + if len(r.Mismatched) != 1 { + t.Fatalf("Mismatched = %v, want one entry", r.Mismatched) + } + if got := r.Mismatched[0]; got.Want != 4096 || got.Got != 400 { + t.Errorf("Mismatch = %+v, want want=4096 got=400", got) + } + if !slices.Contains(r.Todo(), 4242) { + t.Errorf("Todo = %v, want it to include 4242", r.Todo()) + } + if r.Complete() { + t.Error("a truncated object was counted as a complete archive") + } + // It is fetchable, unlike an unsafe name — re-uploading overwrites it. + if !slices.Contains(r.Fetchable(), 4242) { + t.Errorf("Fetchable = %v, want it to include 4242", r.Fetchable()) + } + if r.Stalled() { + t.Error("a repairable file must not be reported as stalled") + } +} From 7382641598eb28cdc7ffe370c0d07e19eeaeefca Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Sun, 6 Sep 2026 20:44:45 +0700 Subject: [PATCH 09/16] docs: record the new exit code, stricter verification, and the filter fix --- README.md | 30 ++++++++++++++++++++++++++---- 1 file changed, 26 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index 5cc42ae..15b3996 100644 --- a/README.md +++ b/README.md @@ -84,7 +84,7 @@ a chat. | `--threads` | 4 | connections per file | | `--limit` | 2 | files downloading at once | | `--uploads` | 2 | files uploading at once | -| `--min-free` | 5 | stop if the remote has fewer than this many GiB free | +| `--min-free` | 5 | stop if the remote has fewer than this many GiB free, checked before and during the run | | `--limit-items` | 0 | stop after N files; for smoke tests | | `--confirm` | true | re-state each uploaded file to prove its size | | `--takeout` | true | use a takeout session | @@ -95,11 +95,15 @@ a chat. | Code | Meaning | |---|---| | 0 | complete | -| 1 | ran, but files remain | +| 1 | ran, but files remain — run again | | 2 | usage error | -| 3 | remote or Telegram failure | +| 3 | remote or Telegram failure, including a destination that stopped accepting uploads | +| 4 | stalled: files remain, none of which can ever be fetched | | 130 / 143 | interrupted (SIGINT / SIGTERM) | +Only 1 is worth retrying. A driver looping until 0 should stop on anything else: +3 and 4 both mean the next pass would do exactly what this one did. + ## How it works Re-running is the resume path. Each item is checked against a listing of the @@ -111,6 +115,13 @@ off and a completed one downloads nothing. and the same string is used both to ask whether the file is already archived and to write it — so the two can never disagree. +A name that cannot survive that round trip is refused rather than rewritten: too +long for the filesystem once `.part` is appended, not a single path element, or +containing a character rclone's path encoder rewrites (control bytes, `DEL`, and +the encoder's own escape character). Those files are reported under +`unarchivable` and never counted as present. Rewriting them is what the next +paragraph is about. + That last point is the reason this program exists. Its predecessor derived the name twice: `tdl chat export` wrote the raw name into a JSON, while `tdl dl` rendered it through a template applying `filenamify`, which rewrites characters a @@ -130,7 +141,14 @@ cap that does not is refused at startup rather than discovered as a hang. **Integrity.** A download is written to `.part` and renamed only once its size matches what Telegram reported, so a file without the suffix is always whole. Uploads are re-stated afterwards to prove they arrived at the right size, -before the local copy is gone. +before the local copy is gone, and an object that turns out short is deleted +rather than left under a name a later run would trust. + +`verify` compares stored sizes against what Telegram reports, so a truncated +object is outstanding rather than "present". This is stricter than the shell +verifier, which matched on name and non-zero size — on the archive this was +built for it found six objects that had been counted complete for months, one +of them 221 MiB standing in for a 2 GiB video. Re-running repairs them. ## Replacing the shell pipeline @@ -154,6 +172,10 @@ semaphore. Some hard-won details were worth keeping, and are: - **Zero-byte files count as missing** — rclone overwrites a size-mismatched destination, so re-running repairs them — while files under 1 KiB are reported but trusted, since some real media genuinely is that small. +- **Indexing ignores `RCLONE_*` filters.** The transfer tunables above are + deliberately env-overridable; the listing is not. A stray `RCLONE_EXCLUDE` or + `RCLONE_MIN_SIZE` left over from another job would otherwise narrow the index + and re-download everything it hid. Flags that disappeared are recognised and explain what replaced them: From 6538b0b9ab6abeb6cd38467b2b31d64d15eb1695 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Sun, 6 Sep 2026 20:51:08 +0700 Subject: [PATCH 10/16] chore: retire the shell pipeline MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit run.sh, export-until-complete.sh and verify-export.sh are replaced by the tgexport binary. Everything expensive in them existed because tdl and rclone could not see each other's state — a staging directory polled with du, an age guard to guess when a download had finished, SIGSTOP/SIGCONT to enforce the disk cap, and an outer loop re-narrowing a JSON export between passes. One process needs none of it. The README keeps the comparison, including the pikpak-specific tunables and the size judgements worth carrying over. --- .gitignore | 12 +- export-until-complete.sh | 113 ------------- run.sh | 333 --------------------------------------- verify-export.sh | 114 -------------- 4 files changed, 6 insertions(+), 566 deletions(-) delete mode 100755 export-until-complete.sh delete mode 100755 run.sh delete mode 100755 verify-export.sh diff --git a/.gitignore b/.gitignore index 42a7b6a..c890cdd 100644 --- a/.gitignore +++ b/.gitignore @@ -1,12 +1,12 @@ -# tdl chat exports and the narrowed subsets built from them. These hold real -# message ids, file names and text from an account, and are specific to one run. -# The repo ships no JSON of its own, so the whole extension is excluded. +# Leftovers from the retired shell pipeline: chat exports and the narrowed +# subsets built from them. These hold real message ids, file names and text from +# an account. The Go binary reads the chat live and writes none of them, but the +# files may still be sitting in a working copy. The repo ships no JSON of its +# own, so the whole extension is excluded. *.json - -# Ids still to fetch, written by verify-export.sh. missing-ids.txt -# Staging holds media in flight between tdl and the remote. +# Staging holds media in flight between the download and upload legs. staging/ # Run logs carry chat names and progress output. diff --git a/export-until-complete.sh b/export-until-complete.sh deleted file mode 100755 index 3095f32..0000000 --- a/export-until-complete.sh +++ /dev/null @@ -1,113 +0,0 @@ -#!/usr/bin/env bash -# -# Drive run.sh repeatedly until every media file in a chat is on the remote. -# -# Each pass verifies what is already there, narrows the export JSON to just the -# ids still needed, and runs the pipeline on that subset. It stops when the -# verifier reports complete, when a pass makes no progress (the remaining media -# is genuinely unavailable on Telegram's side), or when the remote runs low on -# space. -# -# Usage: ./export-until-complete.sh -r REMOTE:PATH -c CHAT [options] -# -r REMOTE:PATH rclone destination (required) -# -c CHAT chat id/username, used for the initial metadata export -# -f FILE export JSON (default export-.json) -# -d DIR staging directory (default ./staging) -# -i SECONDS rclone sweep interval (default 120) -# -p N maximum passes (default 30) -# -m SIZE cap the staging directory at SIZE, e.g. 40G (default: no cap) -# -q GIB stop if remote free space falls below this (default 5) - -set -euo pipefail - -remote='' chat='' export_file='' staging='./staging' interval=120 max_staging='' -max_passes=30 min_free_gib=5 - -while getopts ':r:c:f:d:i:p:q:m:h' o; do case $o in - r) remote=$OPTARG ;; c) chat=$OPTARG ;; f) export_file=$OPTARG ;; - d) staging=$OPTARG ;; i) interval=$OPTARG ;; p) max_passes=$OPTARG ;; - q) min_free_gib=$OPTARG ;; m) max_staging=$OPTARG ;; - h) sed -n '2,19p' "$0"; exit 0 ;; - *) echo "usage: $0 -r REMOTE:PATH -c CHAT [-f FILE] [-i SECS] [-p N] [-q GIB] [-m SIZE]" >&2; exit 2 ;; -esac; done - -log() { printf '\n=== %s [driver] %s ===\n' "$(date '+%Y-%m-%d %H:%M:%S')" "$*"; } - -[[ -n $remote ]] || { echo "-r REMOTE:PATH is required" >&2; exit 2; } -[[ -n $chat || -n $export_file ]] || { echo "-c CHAT or -f FILE is required" >&2; exit 2; } -[[ -n $export_file ]] || export_file="export-${chat}.json" - -# Stop the whole loop on Ctrl-C rather than rolling into the next pass. run.sh -# installs its own handlers, so a pass shuts down cleanly before we exit. -interrupted=0 -trap 'interrupted=1' INT TERM - -# tdl's progress bar is ANSI redraws: great on a terminal, unreadable in a log. -# Show it when stdout is a TTY, suppress it when output is redirected. -tdl_quiet=() -[[ -t 1 ]] || tdl_quiet=(--disable-progress-ps) - -# The metadata export must exist before the first verify has anything to compare. -if [[ ! -f $export_file ]]; then - [[ -n $chat ]] || { echo "$export_file missing and no -c CHAT to create it" >&2; exit 2; } - log "exporting $chat metadata to $export_file" - tdl chat export -c "$chat" --all --with-content -o "$export_file" -fi - -free_gib() { - rclone about "${remote%%:*}:" --json 2>/dev/null \ - | python3 -c 'import json,sys; print(int(json.load(sys.stdin).get("free",0))//2**30)' 2>/dev/null \ - || echo 999999 # backends without quota reporting must not block the run -} - -prev_todo=-1 -for ((pass = 1; pass <= max_passes; pass++)); do - ((interrupted)) && { log 'interrupted; stopping'; exit 130; } - - log "pass $pass/$max_passes: verifying" - if ./verify-export.sh -f "$export_file" -r "$remote" -d "$staging"; then - log "complete after $((pass - 1)) download pass(es)" - exit 0 - fi - - todo=$(wc -l < missing-ids.txt) - if ((todo == prev_todo)); then - log "no progress in the last pass; $todo file(s) look permanently unavailable" - log 'ids left in missing-ids.txt' - exit 1 - fi - prev_todo=$todo - - free=$(free_gib) - if ((free < min_free_gib)); then - log "remote has only ${free} GiB free (limit ${min_free_gib}); stopping before it fills" - exit 3 - fi - log "$todo file(s) to fetch; ${free} GiB free on remote" - - # Narrow the full export to the outstanding ids, preserving its top-level shape - # so tdl reads it exactly like the original. - python3 - "$export_file" <<'PY' -import json, sys -d = json.load(open(sys.argv[1])) -want = {int(l) for l in open('missing-ids.txt') if l.strip()} -d['messages'] = [m for m in d['messages'] if m['id'] in want] -json.dump(d, open('gap.json', 'w')) -print(f"gap.json: {len(d['messages'])} messages") -PY - - rc=0 - ./run.sh -r "$remote" -f gap.json -d "$staging" -i "$interval" \ - ${max_staging:+-m "$max_staging"} \ - -- --group=false -t 4 -l 2 ${tdl_quiet[@]+"${tdl_quiet[@]}"} || rc=$? - log "pass $pass finished (run.sh exit $rc)" - - case $rc in - 0|1) ;; # done or partial: verify decides - 3) log 'rclone failure (remote full or unreachable); stopping'; exit 3 ;; - 130|143) log 'run interrupted; stopping'; exit 130 ;; - esac -done - -log "hit the $max_passes-pass limit; run again to continue" -exit 1 diff --git a/run.sh b/run.sh deleted file mode 100755 index b552068..0000000 --- a/run.sh +++ /dev/null @@ -1,333 +0,0 @@ -#!/usr/bin/env bash -# -# Rolling pipeline: tdl downloads Telegram media into a small staging directory -# while rclone concurrently moves finished files to any rclone remote (S3, -# Google Drive, SFTP, WebDAV, B2, ...). Local disk only ever holds the files in -# flight plus one sync interval of throughput, so a chat larger than the local -# disk can still be exported. -# -# See README.md for the background. -# -# Exit codes: 0 ok, 2 usage error, 3 rclone failure, 130/143 interrupted, -# anything else is tdl's own exit code. - -set -euo pipefail - -readonly PROG=${0##*/} - -# tdl downloads to '.tmp' and renames only on completion, so an unfinished -# file is always identifiable by extension. Never move one: a download stalled -# by a flood wait stops touching its .tmp, which then ages past --min-age and -# would be uploaded half-written, destroying tdl's resume point for that file. -readonly TEMP_GLOB='*.tmp' - -# Defaults -export_file='' # decided after parsing: per-chat when -c is given -chat='' -staging='./staging' -remote='' -interval=60 -min_age='45s' -max_staging='' # empty: staging grows as fast as tdl fills it -max_sync_failures=5 -cap_check_interval=10 # seconds between -m checks, independent of -i - -usage() { - cat <.json with -c, - otherwise export.json) - -c CHAT chat to export when FILE does not exist. Accepts a numeric - id as printed by 'tdl chat ls', a username with or without - '@', or a t.me/tg:// link. A Bot API '-100...' id is - converted to the plain id tdl expects. - -d DIR staging directory (default: $staging) - -i SECONDS seconds between rclone sweeps (default: $interval) - -a AGE rclone --min-age, a second guard against moving files still - being written (default: $min_age) - -m SIZE cap the staging directory at SIZE (K/M/G/T, binary; e.g. - 40G). When staging reaches it, tdl is suspended until - rclone has drained the finished files. Unset means no cap. - -h this help - -Everything after -- is appended to the 'tdl dl' command, e.g. - $PROG -r gdrive:telegram/media -- -t 4 -l 1 -USAGE -} - -log() { printf '%s [%s] %s\n' "$(date '+%Y-%m-%d %H:%M:%S')" "$PROG" "$*" >&2; } -die() { log "error: $*"; exit 2; } # usage or precondition -fail() { log "error: $*"; exit 3; } # rclone / pipeline failure - -while getopts ':f:c:d:r:i:a:m:h' opt; do - case $opt in - f) export_file=$OPTARG ;; - c) chat=$OPTARG ;; - d) staging=$OPTARG ;; - r) remote=$OPTARG ;; - i) interval=$OPTARG ;; - a) min_age=$OPTARG ;; - m) max_staging=$OPTARG ;; - h) usage; exit 0 ;; - :) die "option -$OPTARG requires an argument" ;; - ?) die "unknown option -$OPTARG (try -h)" ;; - esac -done -shift $((OPTIND - 1)) -tdl_extra=("$@") - -[[ -n $remote ]] || { usage >&2; die "-r REMOTE:PATH is required"; } -[[ $remote == *:* ]] || die "remote '$remote' is not in rclone REMOTE:PATH form" -[[ $interval =~ ^[0-9]+$ && $interval -gt 0 ]] || die "-i must be a positive integer" -[[ $min_age =~ ^[0-9]+(\.[0-9]+)?(ms|s|m|h|d|w|M|y)?$ ]] \ - || die "-a must be an rclone duration, e.g. 2m" - -# Sizes are compared in KiB because that is the unit 'du -sk' reports, which is -# also the unit that matters here: allocated blocks, not apparent length. -max_staging_kib='' -if [[ -n $max_staging ]]; then - [[ $max_staging =~ ^[0-9]+[KkMmGgTt]?$ ]] \ - || die "-m must be a size with an optional K/M/G/T suffix, e.g. 40G" - num=${max_staging%[KkMmGgTt]} - case ${max_staging#"$num"} in - ''|K|k) max_staging_kib=$num ;; - M|m) max_staging_kib=$((num * 1024)) ;; - G|g) max_staging_kib=$((num * 1024 * 1024)) ;; - T|t) max_staging_kib=$((num * 1024 * 1024 * 1024)) ;; - esac - ((max_staging_kib > 0)) || die "-m must be greater than zero" -fi - -# tdl resolves a numeric argument as an MTProto id and anything else through -# gotd's resolver, which handles '@name', 'name' and t.me/tg:// links. Two forms -# still need help: Bot API ids carry a '-100' prefix that MTProto does not use, -# and a message link is not a chat. -if [[ -n $chat ]]; then - read -r chat <<<"$chat" # trim stray whitespace - case $chat in - '') die "-c requires a chat" ;; - *t.me/c/*|*t.me/*/[0-9]*) - die "-c takes a chat, not a message link ($chat) — pass the chat's username or id" ;; - -100[0-9]*) - log "converting Bot API id $chat to MTProto id ${chat#-100}" - chat=${chat#-100} ;; - esac -fi - -# Keep each chat's export in its own file, so switching -c never silently -# downloads the previous chat again from a stale export.json. -if [[ -z $export_file ]]; then - if [[ -n $chat ]]; then - slug=${chat#@} - slug=${slug##*/} - slug=$(printf '%s' "$slug" | tr -c 'A-Za-z0-9._-' '_') - export_file="export-$slug.json" - else - export_file='export.json' - fi -fi - -# pikpak-style backends finish an upload as a server-side async task, and -# rclone abandons a still-pending one once --low-level-retries polls run out, -# failing a transfer that would have succeeded. Fewer parallel transfers keep -# that queue short and more retries wait it out. These are rclone's own -# environment variables, so whatever the caller already exported wins. -: "${RCLONE_TRANSFERS:=2}" -: "${RCLONE_LOW_LEVEL_RETRIES:=20}" -export RCLONE_TRANSFERS RCLONE_LOW_LEVEL_RETRIES - -for tool in tdl rclone; do - command -v "$tool" >/dev/null || die "$tool is not installed or not on PATH" -done - -# A named remote must exist in the config; a leading ':' means an on-the-fly -# connection string, which has no config entry to check. Listing the remote's -# root is not portable (some backends refuse it), so reachability and -# credentials are proven by creating the destination, which rclone would create -# on the first move anyway. -if [[ $remote != :* ]]; then - # Read the list into a variable first: piping it into 'grep -q' lets grep exit - # on the first match and kill rclone with SIGPIPE, which pipefail then reports - # as a failed pipeline -- rejecting a remote that is in fact configured. - remotes=$(rclone listremotes 2>/dev/null || true) - grep -qx -- "${remote%%:*}:" <<<"$remotes" \ - || die "rclone remote '${remote%%:*}:' is not configured — see 'rclone listremotes'" -fi -rclone mkdir "$remote" >/dev/null 2>&1 \ - || die "cannot reach '$remote' — check credentials and connectivity" - -# The export JSON only lists messages; it is cheap to keep and required for both -# legs to stay resumable, so never regenerate it when it already exists. -if [[ ! -f $export_file ]]; then - [[ -n $chat ]] || die "$export_file not found; pass -c CHAT to export it first" - log "exporting $chat metadata to $export_file" - tdl chat export -c "$chat" --all --with-content -o "$export_file" -else - log "using existing $export_file (delete it to re-export)" -fi - -mkdir -p "$staging" - -tdl_pid='' -tdl_rc=0 -sweep_ok=0 - -# The sweeps that run once at the end have a whole staging directory to move -# and no interleaved tdl output, so they report progress instead of going quiet -# for several minutes. rclone's redrawn bar is right on a terminal but turns a -# redirected log into control characters, so a log gets periodic one-line -# stats. Those are logged at INFO, which '-v' would enable at the cost of a -# line per file, hence raising the stats to NOTICE rather than the whole log. -sweep_progress=(--progress) -[[ -t 1 ]] || sweep_progress=(--stats 30s --stats-one-line --stats-log-level NOTICE) - -# Move whatever is finished. Partial .tmp files are always excluded. $1 is an -# optional --min-age guard; $2 enables --delete-empty-src-dirs, which is safe -# only once tdl has stopped — rclone removing a directory between tdl's -# MkdirAll and Create makes tdl fail, and it can take the staging root too, -# hence the mkdir afterwards. $3 turns on the progress reporting above. -sweep() { - local rc=0 - local args=(--exclude "$TEMP_GLOB") - if [[ -n ${1:-} ]]; then args+=(--min-age "$1"); fi - if ((${2:-0})); then args+=(--delete-empty-src-dirs); fi - if ((${3:-0})); then args+=(${sweep_progress[@]+"${sweep_progress[@]}"}); fi - rclone move "$staging" "$remote" "${args[@]}" || rc=$? - mkdir -p "$staging" - return $rc -} - -staging_kib() { du -sk "$staging" 2>/dev/null | cut -f1; } - -over_cap() { - local used - used=$(staging_kib) - [[ -n $used ]] && ((used >= max_staging_kib)) -} - -# Enforce -m. tdl renames a file only once it is complete, so staging holds -# finished files plus the in-flight '*.tmp' ones, and only the former can be -# drained. Suspending tdl stops it adding more while rclone empties the -# directory; SIGSTOP is safe because tdl reconnects on resume and --continue -# picks its .tmp files back up. The cap must therefore stay well above what the -# concurrent downloads hold, or draining could never clear it. -drain_to_cap() { - local used rc=0 - used=$(staging_kib) - log "staging at $((used / 1024)) MiB, at or over the $((max_staging_kib / 1024)) MiB cap; suspending tdl to drain" - - kill -STOP "$tdl_pid" 2>/dev/null || true - # Sweep until staging is back under the cap, since one rclone move need not - # get there: a slow remote or a per-file error can leave finished files - # behind. Stop as soon as a sweep frees nothing, which means all that is - # left is in-flight .tmp files that no sweep can ever move. - local before - while :; do - before=$used - # No --min-age: tdl is frozen, so every non-.tmp file is finished by - # construction and waiting out the guard would only prolong the pause. - sweep '' || { rc=$?; break; } - used=$(staging_kib) - [[ -n $used ]] || break - ((used >= max_staging_kib && used < before)) || break - done - kill -CONT "$tdl_pid" 2>/dev/null || true - - if ((rc == 0)) && [[ -n $used ]] && ((used >= max_staging_kib)); then - log "warning: staging is still $((used / 1024)) MiB after draining — the cap is below what the in-flight downloads hold; raise -m or lower tdl's -l/-t" - else - log "resumed tdl, staging now $((${used:-0} / 1024)) MiB" - fi - return $rc -} - -cleanup() { - if [[ -n $tdl_pid ]] && kill -0 "$tdl_pid" 2>/dev/null; then - log "stopping tdl (pid $tdl_pid)" - # A suspended process never sees SIGTERM, so let it run first. - kill -CONT "$tdl_pid" 2>/dev/null || true - kill -TERM "$tdl_pid" 2>/dev/null || true - for _ in 1 2 3 4 5 6 7 8 9 10; do - kill -0 "$tdl_pid" 2>/dev/null || break - sleep 1 - done - kill -KILL "$tdl_pid" 2>/dev/null || true - fi - if ((sweep_ok)); then - log 'sweeping completed files before exit' - sweep "$min_age" 0 1 || log 'warning: final safety sweep failed; staging kept' - fi -} -trap cleanup EXIT -trap 'log "interrupted (SIGINT)"; exit 130' INT -trap 'log "terminated (SIGTERM)"; exit 143' TERM - -log "downloading into $staging, moving to $remote every ${interval}s${max_staging_kib:+, capped at $((max_staging_kib / 1024)) MiB}" -tdl dl -f "$export_file" -d "$staging" \ - --takeout --group --skip-same --continue ${tdl_extra[@]+"${tdl_extra[@]}"} & -tdl_pid=$! -sweep_ok=1 - -failures=0 -waited=0 -while kill -0 "$tdl_pid" 2>/dev/null; do - # A long sweep interval must not let staging blow past the cap in between, so - # sleep in slices and check the cap on each one. Every sleep is a job, so - # signals are handled without waiting the slice out. - slice=$((interval - waited)) - if [[ -n $max_staging_kib ]] && ((slice > cap_check_interval)); then - slice=$cap_check_interval - fi - sleep "$slice" & - wait $! 2>/dev/null || true - waited=$((waited + slice)) - - # Only a sweep that actually ran says anything about rclone's health, so the - # failure streak is judged on those alone and a quiet cap check never - # clears it. - rc=0 - swept=0 - if ((waited >= interval)); then - waited=0 - swept=1 - sweep "$min_age" || rc=$? - fi - if ((rc == 0)) && [[ -n $max_staging_kib ]] && kill -0 "$tdl_pid" 2>/dev/null; then - if over_cap; then - swept=1 - drain_to_cap || rc=$? - fi - fi - - if ((swept)); then - if ((rc == 0)); then - failures=0 - else - failures=$((failures + 1)) - log "warning: rclone sweep failed ($failures/$max_sync_failures)" - if ((failures >= max_sync_failures)); then - sweep_ok=0 - fail "rclone failed $failures times in a row; stopping before staging fills the disk" - fi - fi - fi -done - -wait "$tdl_pid" || tdl_rc=$? -tdl_pid='' -sweep_ok=0 - -if ((tdl_rc != 0)); then - log "tdl exited $tdl_rc; staging kept at $staging — re-run to resume" - sweep "$min_age" 0 1 || log 'warning: sweep after failure did not complete' - exit "$tdl_rc" -fi - -log 'tdl finished; final sweep' -sweep '' 1 1 || fail "final sweep failed; files remain in $staging" -log "done — everything moved to $remote" diff --git a/verify-export.sh b/verify-export.sh deleted file mode 100755 index 064d123..0000000 --- a/verify-export.sh +++ /dev/null @@ -1,114 +0,0 @@ -#!/usr/bin/env bash -# -# Verify a tdl/rclone export is complete and every file is intact. -# -# Rebuilds the exact filename tdl produces for each media message in the export -# JSON ({DialogID}_{MessageID}_{FileName}) and checks it exists on the remote or -# in staging. The export is Telegram's own record of the name, so the match is -# exact: a file stored under any other name is not the file the export asked -# for and counts as absent, however close the name looks. Zero-byte files count -# as missing too: rclone overwrites a size-mismatched destination, so re-running -# repairs them. Files under 1 KiB are reported for inspection but trusted, since -# some real media is genuinely that small. -# -# A remote file whose id matches but whose name does not is a stale copy from an -# earlier download; it is listed separately so it can be deleted, because the -# re-download lands beside it rather than replacing it. -# -# Writes every id needing another attempt to missing-ids.txt. -# Exit 0 = complete, 1 = incomplete, 2 = usage error. -# -# Usage: ./verify-export.sh -f export-.json -r REMOTE:PATH [-d STAGING] - -set -euo pipefail - -export_file='' remote='' staging='./staging' -while getopts ':f:r:d:h' o; do case $o in - f) export_file=$OPTARG ;; r) remote=$OPTARG ;; d) staging=$OPTARG ;; - h) sed -n '2,16p' "$0"; exit 0 ;; - *) echo "usage: $0 -f FILE -r REMOTE:PATH [-d STAGING]" >&2; exit 2 ;; -esac; done - -[[ -f $export_file ]] || { echo "no such export file: $export_file" >&2; exit 2; } -[[ -n $remote ]] || { echo "-r REMOTE:PATH is required" >&2; exit 2; } - -listing=$(mktemp); trap 'rm -f "$listing"' EXIT -# lsl gives sizes as well as names, so truncated uploads are detectable. -rclone lsl "$remote" > "$listing" - -python3 - "$export_file" "$listing" "$staging" <<'PY' -import json, os, re, sys -export_file, listing, staging = sys.argv[1], sys.argv[2], sys.argv[3] - -sizes = {} -for line in open(listing): - m = re.match(r'^\s*(\d+)\s+\S+\s+\S+\s+(.*)$', line.rstrip('\n')) - if m: - sizes[os.path.basename(m.group(2))] = int(m.group(1)) -if os.path.isdir(staging): - for f in os.listdir(staging): - if not f.endswith('.tmp'): - sizes.setdefault(f, os.path.getsize(os.path.join(staging, f))) - -msgs = json.load(open(export_file))['messages'] -# Text-only messages carry an empty "file" and are not download targets. -media = [m for m in msgs if m.get('file')] - -dialog = str(json.load(open(export_file)).get('id', '')).lstrip('-') -if not dialog.isdigit(): - ids = {n.split('_')[0] for n in sizes if '_' in n} - dialog = ids.pop() if len(ids) == 1 else '' -if not dialog: - sys.exit('cannot determine dialog id') - -# Names already stored for each message id, used only to tell an absent file -# apart from one sitting there under the wrong name. -stored = {} -for name in sizes: - parts = name.split('_', 2) - if len(parts) == 3 and parts[0] == dialog and parts[1].isdigit(): - stored.setdefault(int(parts[1]), []).append(name) - -absent, empty, tiny, misnamed = [], [], [], [] -for m in media: - name = f"{dialog}_{m['id']}_{m['file']}" - if name not in sizes: - absent.append(m['id']) - for other in stored.get(m['id'], []): - misnamed.append((m['id'], name, other, sizes[other])) - elif sizes[name] == 0: - empty.append(m['id']) - elif sizes[name] < 1024: - tiny.append((m['id'], sizes[name], name)) - -todo = sorted(absent + empty) -print(f"messages in export : {len(msgs)}") -print(f" text-only (skip) : {len(msgs) - len(media)}") -print(f" media expected : {len(media)}") -print(f"present and intact : {len(media) - len(todo)}") -print(f" absent : {len(absent)}") -print(f" zero-byte : {len(empty)}") -if tiny: - print(f" under 1KiB (check, not retried): {len(tiny)}") - for i, s, n in tiny[:5]: - print(f" id {i} {s} B {n}") -if misnamed: - print(f"\nstored under a different name : {len(misnamed)}") - print(" counted as absent and fetched again; delete the stale copies so the") - print(" re-download does not leave two files for the same message:") - for i, want, got, size in misnamed: - print(f" id {i} {size} B") - print(f" export: {want}") - print(f" remote: {got}") - -if todo: - with open('missing-ids.txt', 'w') as fh: - fh.write('\n'.join(map(str, todo)) + '\n') - print(f"\nneeds another pass : {len(todo)} (ids {min(todo)}–{max(todo)})") - print("written to missing-ids.txt") - sys.exit(1) - -if os.path.exists('missing-ids.txt'): - os.remove('missing-ids.txt') -print("\nCOMPLETE: every media message is present and non-empty.") -PY From fafeab47500b40ad26f9dc9bd48ea31befb80281 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Sun, 6 Sep 2026 21:05:59 +0700 Subject: [PATCH 11/16] feat: report progress while reading a chat and listing a remote MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Both phases ran for minutes printing nothing between their opening line and their result, so a working run looked exactly like a hung one — which is how it was reported. The message counter lives in Walk rather than in the caller's loop because most of a chat is not media: text-only and service messages are filtered out inside the walk, so a caller counting yielded items still sees nothing while crossing a long stretch of conversation. Cadence follows the existing reporter: a terminal redraws one line, a redirected run gets a periodic one, since ANSI redraws are what turned the shell pipeline's captured logs into megabytes of control characters. --- cmd/tgexport/list.go | 4 +- cmd/tgexport/sync.go | 28 ++++++++-- cmd/tgexport/verify.go | 34 ++++++++---- internal/remote/index.go | 9 +++- internal/remote/index_env_test.go | 2 +- internal/remote/index_test.go | 2 +- internal/report/ticker.go | 89 +++++++++++++++++++++++++++++++ internal/report/ticker_test.go | 59 ++++++++++++++++++++ internal/tgsource/iterate.go | 14 ++++- 9 files changed, 222 insertions(+), 19 deletions(-) create mode 100644 internal/report/ticker.go create mode 100644 internal/report/ticker_test.go diff --git a/cmd/tgexport/list.go b/cmd/tgexport/list.go index 81981af..5cdccb0 100644 --- a/cmd/tgexport/list.go +++ b/cmd/tgexport/list.go @@ -11,6 +11,7 @@ import ( "github.com/iyear/tdl/core/dcpool" tdlstorage "github.com/iyear/tdl/core/storage" + "github.com/tiennm99dev/telegram-exporter/internal/report" "github.com/tiennm99dev/telegram-exporter/internal/tdlkv" "github.com/tiennm99dev/telegram-exporter/internal/tgsource" ) @@ -66,9 +67,10 @@ func listCmd(ctx context.Context, args []string) error { // swallowed by a deferred call nobody checks. out := bufio.NewWriter(os.Stdout) + scan := report.NewTicker(os.Stderr, "messages read") count := 0 var total int64 - for it, err := range tgsource.Walk(ctx, api, peer) { + for it, err := range tgsource.Walk(ctx, api, peer, scan.Update) { if err != nil { return err } diff --git a/cmd/tgexport/sync.go b/cmd/tgexport/sync.go index b87f6c1..6b3a6a1 100644 --- a/cmd/tgexport/sync.go +++ b/cmd/tgexport/sync.go @@ -122,15 +122,21 @@ func syncCmd(ctx context.Context, args []string) error { } fmt.Fprintf(os.Stderr, "reading %s\n", *chat) + scan := report.NewTicker(os.Stderr, "messages read") var items []tgsource.Item - for it, err := range tgsource.Walk(ctx, api, peer) { + var scanned int + for it, err := range tgsource.Walk(ctx, api, peer, func(n int) { + scanned = n + scan.Update(n) + }) { if err != nil { return err } items = append(items, it) } + scan.Done(scanned) - idx, err := remote.BuildIndex(ctx, dst, peer.ID()) + idx, err := indexRemote(ctx, dst, peer.ID()) if err != nil { return err } @@ -191,7 +197,7 @@ func syncCmd(ctx context.Context, args []string) error { // The remote is re-indexed rather than assumed: the run's own view of // what it uploaded is exactly the thing under test. - idx, err = remote.BuildIndex(ctx, dst, peer.ID()) + idx, err = indexRemote(ctx, dst, peer.ID()) if err != nil { return err } @@ -280,6 +286,22 @@ func validateBudget(budget int64, todo []tgsource.Item) error { return nil } +// indexRemote lists the destination, reporting progress as it goes. +func indexRemote(ctx context.Context, dst fs.Fs, dialogID int64) (*remote.Index, error) { + fmt.Fprintf(os.Stderr, "indexing %s\n", dst.String()) + tick := report.NewTicker(os.Stderr, "objects listed") + var seen int + idx, err := remote.BuildIndex(ctx, dst, dialogID, func(n int) { + seen = n + tick.Update(n) + }) + if err != nil { + return nil, err + } + tick.Done(seen) + return idx, nil +} + // warnCollisions reports basenames the index found at more than one path. // // verify printed this and sync did not, which was backwards: an ambiguous diff --git a/cmd/tgexport/verify.go b/cmd/tgexport/verify.go index d79d1e7..8239e47 100644 --- a/cmd/tgexport/verify.go +++ b/cmd/tgexport/verify.go @@ -15,6 +15,7 @@ import ( "github.com/rclone/rclone/fs/operations" "github.com/tiennm99dev/telegram-exporter/internal/remote" + "github.com/tiennm99dev/telegram-exporter/internal/report" "github.com/tiennm99dev/telegram-exporter/internal/tdlkv" "github.com/tiennm99dev/telegram-exporter/internal/tgsource" "github.com/tiennm99dev/telegram-exporter/internal/verify" @@ -66,7 +67,7 @@ func verifyCmd(ctx context.Context, args []string) error { } var ( - report verify.Report + result verify.Report idx *remote.Index ) if err := sess.Run(ctx, func(ctx context.Context, pool dcpool.Pool) error { @@ -80,19 +81,30 @@ func verifyCmd(ctx context.Context, args []string) error { // The chat is walked first and the remote listed second, so the snapshot // is never older than the wanted set. The reverse order could report a // file absent that was uploaded while the walk was still running. + fmt.Fprintf(os.Stderr, "reading %s\n", *chat) + scan := report.NewTicker(os.Stderr, "messages read") + var scanned int var items []tgsource.Item - for it, err := range tgsource.Walk(ctx, api, peer) { + for it, err := range tgsource.Walk(ctx, api, peer, func(n int) { scanned = n; scan.Update(n) }) { if err != nil { return err } items = append(items, it) } - idx, err = remote.BuildIndex(ctx, dst, peer.ID()) + scan.Done(scanned) + + fmt.Fprintf(os.Stderr, "indexing %s\n", dst.String()) + idxTick := report.NewTicker(os.Stderr, "objects listed") + var listed int + idx, err = remote.BuildIndex(ctx, dst, peer.ID(), func(n int) { + listed = n + idxTick.Update(n) + }) if err != nil { return err } - fmt.Fprintf(os.Stderr, "indexed %d objects on %s\n", idx.Len(), dst.String()) + idxTick.Done(listed) if dup := idx.Collisions(); len(dup) > 0 { // An ambiguous snapshot makes every verdict about these names // unreliable, so it is reported rather than silently resolved. @@ -104,26 +116,26 @@ func verifyCmd(ctx context.Context, args []string) error { } fmt.Fprintln(os.Stderr) - report = verify.Check(items, idx) + result = verify.Check(items, idx) return nil }); err != nil { return err } out := bufio.NewWriter(os.Stdout) - report.Write(out) + result.Write(out) if err := out.Flush(); err != nil { return err } if *delStale { - if err := deleteMisnamed(ctx, dst, idx, report, *assumeYes); err != nil { + if err := deleteMisnamed(ctx, dst, idx, result, *assumeYes); err != nil { return err } } - if !report.Complete() { - return fmt.Errorf("%w: %d file(s) still to fetch", errIncomplete, len(report.Todo())) + if !result.Complete() { + return fmt.Errorf("%w: %d file(s) still to fetch", errIncomplete, len(result.Todo())) } return nil } @@ -138,9 +150,9 @@ func verifyCmd(ctx context.Context, args []string) error { // used elsewhere. Those differ the moment a remote has directory structure, and // deleting by basename would either miss the object or — worse, if the root // happens to hold a same-named file — delete the wrong one. -func deleteMisnamed(ctx context.Context, dst rclonefs.Fs, idx *remote.Index, report verify.Report, assumeYes bool) error { +func deleteMisnamed(ctx context.Context, dst rclonefs.Fs, idx *remote.Index, result verify.Report, assumeYes bool) error { var targets []string - for _, m := range report.Misnamed { + for _, m := range result.Misnamed { targets = append(targets, m.Found...) } if len(targets) == 0 { diff --git a/internal/remote/index.go b/internal/remote/index.go index 4e97e5c..c95af98 100644 --- a/internal/remote/index.go +++ b/internal/remote/index.go @@ -62,7 +62,9 @@ type Index struct { // why every field that can narrow a listing is set explicitly. A zero-value // Options is not a substitute either — it fails validation, because MinAge and // MaxAge both being 0 reads as "min > max". -func BuildIndex(ctx context.Context, f fs.Fs, dialogID int64) (*Index, error) { +// onCount, when non-nil, is called with the number of objects seen so far. +// Listing a remote of any size takes minutes and says nothing while it runs. +func BuildIndex(ctx context.Context, f fs.Fs, dialogID int64, onCount func(n int)) (*Index, error) { ctx, ci := fs.AddConfig(ctx) ci.MaxDepth = -1 @@ -88,7 +90,12 @@ func BuildIndex(ctx context.Context, f fs.Fs, dialogID int64) (*Index, error) { } // ListFn is documented not to call fn concurrently, so the maps need no lock. + seen := 0 if err := operations.ListFn(ctx, f, func(o fs.Object) { + seen++ + if onCount != nil { + onCount(seen) + } name := path.Base(o.Remote()) if _, seen := idx.byName[name]; seen { diff --git a/internal/remote/index_env_test.go b/internal/remote/index_env_test.go index 1fb4124..d4a05a3 100644 --- a/internal/remote/index_env_test.go +++ b/internal/remote/index_env_test.go @@ -52,7 +52,7 @@ func indexChild(t *testing.T) { if err != nil { t.Fatal(err) } - idx, err := BuildIndex(t.Context(), f, 1234567890) + idx, err := BuildIndex(t.Context(), f, 1234567890, nil) if err != nil { t.Fatalf("BuildIndex: %v", err) } diff --git a/internal/remote/index_test.go b/internal/remote/index_test.go index 5abd8c4..eeb1b15 100644 --- a/internal/remote/index_test.go +++ b/internal/remote/index_test.go @@ -33,7 +33,7 @@ func localIndex(t *testing.T, files map[string]int) *Index { if err != nil { t.Fatalf("open local fs: %v", err) } - idx, err := BuildIndex(ctx, f, testDialog) + idx, err := BuildIndex(ctx, f, testDialog, nil) if err != nil { t.Fatalf("BuildIndex: %v", err) } diff --git a/internal/report/ticker.go b/internal/report/ticker.go new file mode 100644 index 0000000..c674521 --- /dev/null +++ b/internal/report/ticker.go @@ -0,0 +1,89 @@ +package report + +import ( + "fmt" + "io" + "sync" + "time" +) + +// tickerInterval is how often a terminal redraws a phase counter. Fast enough +// to look alive, slow enough not to matter. +const tickerInterval = 250 * time.Millisecond + +// Ticker reports progress through a long phase that would otherwise be silent. +// +// Reading a chat's history and listing a remote each take minutes on an archive +// of any size, and both used to print nothing between their opening line and +// their result. A run that is working looked identical to one that had hung, so +// the only way to tell was to wait it out. +// +// Cadence follows the same rule as Reporter: a terminal gets a redrawn line, a +// redirected run gets a periodic one, because ANSI redraws turn a captured log +// into megabytes of control characters. +type Ticker struct { + w io.Writer + tty bool + noun string + start time.Time + + mu sync.Mutex + lastLine time.Time +} + +// NewTicker builds a ticker that counts noun, e.g. "messages" or "objects". +func NewTicker(w io.Writer, noun string) *Ticker { + return &Ticker{w: w, tty: isTerminal(w), noun: noun, start: time.Now()} +} + +// Update reports a running count. Safe to call from several goroutines, and +// cheap enough to call per item. +func (t *Ticker) Update(n int) { + if !t.mu.TryLock() { + return + } + defer t.mu.Unlock() + + now := time.Now() + interval := statsInterval + if t.tty { + interval = tickerInterval + } + if now.Sub(t.lastLine) < interval { + return + } + t.lastLine = now + + if t.tty { + fmt.Fprintf(t.w, "\r\033[K %s %s...", humanCount(n), t.noun) + return + } + fmt.Fprintf(t.w, " %s %s...\n", humanCount(n), t.noun) +} + +// Done clears the redrawn line and states the final count. +func (t *Ticker) Done(n int) { + t.mu.Lock() + defer t.mu.Unlock() + if t.tty { + fmt.Fprint(t.w, "\r\033[K") + } + fmt.Fprintf(t.w, " %s %s in %s\n", humanCount(n), t.noun, + time.Since(t.start).Round(time.Second)) +} + +// humanCount groups thousands, so 12000 reads as 12,000. +func humanCount(n int) string { + s := fmt.Sprintf("%d", n) + if len(s) <= 3 { + return s + } + out := make([]byte, 0, len(s)+len(s)/3) + for i, c := range []byte(s) { + if i > 0 && (len(s)-i)%3 == 0 { + out = append(out, ',') + } + out = append(out, c) + } + return string(out) +} diff --git a/internal/report/ticker_test.go b/internal/report/ticker_test.go new file mode 100644 index 0000000..66b3ffa --- /dev/null +++ b/internal/report/ticker_test.go @@ -0,0 +1,59 @@ +package report + +import ( + "strings" + "testing" + "time" +) + +func TestHumanCountGroupsThousands(t *testing.T) { + cases := map[int]string{ + 0: "0", 7: "7", 999: "999", 1000: "1,000", + 12000: "12,000", 11406: "11,406", 1234567: "1,234,567", + } + for in, want := range cases { + if got := humanCount(in); got != want { + t.Errorf("humanCount(%d) = %q, want %q", in, got, want) + } + } +} + +// A redirected run must not accumulate ANSI redraws: that is what turned the +// shell pipeline's captured logs into megabytes of control characters. +func TestTickerWritesNoAnsiWhenRedirected(t *testing.T) { + var sb strings.Builder + tick := NewTicker(&sb, "messages read") + + for i := 1; i <= 5000; i++ { + tick.Update(i) + } + tick.Done(5000) + + out := sb.String() + if strings.ContainsAny(out, "\r\033") { + t.Errorf("redirected output contains control characters: %q", out) + } + if !strings.Contains(out, "5,000 messages read") { + t.Errorf("final count missing from %q", out) + } +} + +// Update is called once per message on an 18k-message walk, so it has to be +// cheap: at most one line per interval, however often it is called. +func TestTickerThrottlesUpdates(t *testing.T) { + var sb strings.Builder + tick := NewTicker(&sb, "objects listed") + tick.lastLine = time.Now() // inside the interval from the start + + for i := 1; i <= 10000; i++ { + tick.Update(i) + } + if n := strings.Count(sb.String(), "\n"); n != 0 { + t.Errorf("wrote %d lines inside one interval, want 0", n) + } + + tick.Done(10000) + if !strings.Contains(sb.String(), "10,000 objects listed in") { + t.Errorf("Done did not state the total: %q", sb.String()) + } +} diff --git a/internal/tgsource/iterate.go b/internal/tgsource/iterate.go index 87c90db..ae44dbc 100644 --- a/internal/tgsource/iterate.go +++ b/internal/tgsource/iterate.go @@ -47,12 +47,24 @@ func (i Item) Size() int64 { return i.Media.Size } // the export JSON encoded as an empty "file" field. On error the sequence yields // a zero Item with that error and stops; cancelling ctx stops it too, so an // interrupted run does not keep paging. -func Walk(ctx context.Context, api *tg.Client, peer peers.Peer) iter.Seq2[Item, error] { +// onScan, when non-nil, is called with the number of messages read so far. It +// has to live here rather than in the caller's loop because most of a chat is +// not media: text-only and service messages are filtered out below, so a caller +// counting yielded items sees nothing at all while the walk crosses a long +// stretch of conversation, and a working run is indistinguishable from a hung +// one. +func Walk(ctx context.Context, api *tg.Client, peer peers.Peer, onScan func(scanned int)) iter.Seq2[Item, error] { return func(yield func(Item, error) bool) { dialogID := peer.ID() it := query.NewQuery(api).Messages().GetHistory(peer.InputPeer()).BatchSize(100).Iter() + scanned := 0 for it.Next(ctx) { + scanned++ + if onScan != nil { + onScan(scanned) + } + msg, ok := it.Value().Msg.(*tg.Message) if !ok { continue // service messages have no media From 4624c506e50a8d0f2dd0f6457f4d8d5bdeffd383 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Sun, 6 Sep 2026 21:21:07 +0700 Subject: [PATCH 12/16] feat: show per-file progress and state the plan before a run starts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A run reported one aggregate line, which could say how much was done but never what was happening: which files were moving, whether a stall was a slow download or a slow upload, or how long the rest would take. On a transfer measured in hours those are the only questions worth answering. The pipeline now emits a per-item lifecycle rather than only Stats, and a terminal renders it as an overall bar plus one bar per file in flight. Uploads get a spinner rather than a bar because rclone's MoveFile is a single blocking call with no byte callbacks; naming the file is still the point, since a run that looks stalled is usually waiting on one large object. Before the transfer, the run states what it found and what it will do. The survey breaks the outstanding set down by reason — never fetched, zero-byte, wrong size, unarchivable — because that is the difference between a run that will converge and one that cannot, and the old "N to fetch" hid it. The plan states the cap, staging path and concurrency, so a wrong setting is visible before hours of transfer rather than after. Redirected output keeps plain lines and gains one per archived file. Bars are continuous cursor movement, and a captured log of them is what tdl's progress bar did to the shell pipeline's logs. cmd/uidemo renders the whole thing against fake data. It is how the layout was checked without a session, and it earned its place immediately: the overall bar sat at zero because Stats only tracked the file count and never advanced it. --- README.md | 27 ++++ cmd/tgexport/sync.go | 22 +++- cmd/uidemo/main.go | 75 +++++++++++ go.mod | 6 +- go.sum | 10 ++ internal/pipeline/download.go | 8 +- internal/pipeline/events.go | 38 ++++++ internal/pipeline/pipeline.go | 11 +- internal/pipeline/progress.go | 24 ++-- internal/report/live.go | 219 ++++++++++++++++++++++++++++++++ internal/report/progress.go | 99 +++++++++++---- internal/report/summary.go | 61 +++++++++ internal/report/summary_test.go | 95 ++++++++++++++ 13 files changed, 644 insertions(+), 51 deletions(-) create mode 100644 cmd/uidemo/main.go create mode 100644 internal/pipeline/events.go create mode 100644 internal/report/live.go create mode 100644 internal/report/summary.go create mode 100644 internal/report/summary_test.go diff --git a/README.md b/README.md index 15b3996..b85c8f6 100644 --- a/README.md +++ b/README.md @@ -57,6 +57,33 @@ build unless the destination will never change. ./tgexport sync -c CHAT -r REMOTE:PATH [options] ``` +A run states what it found, what it is about to do, and then shows each file as +it moves: + +``` +reading mychannel + 12,000 messages read in 4m31s +indexing PikPak root 'mychannel' + 11,406 objects listed in 2m10s + + chat holds 12,000 media, 250.0 GiB + archived 11,400 + never fetched 600 + wrong size 6 + + fetching 606 files, 79.0 GiB (largest 2.0 GiB) + into PikPak root 'mychannel' + staging ./staging, capped at 40.0 GiB + concurrency 2 download(s) x 4 thread(s), 2 upload(s) + + total 26/606 files [=> ] 617.5 MiB / 79.0 GiB 2.5 MiB/s 8h47m + ↓ …3214_4242_1000000000000000001.mp4 [=======> ] 41.2 MiB / 96.0 MiB 1.8 MiB/s + ↑ …3214_4243_1000000000000000002.mp4 ⠹ uploading 1.9 GiB +``` + +Redirected output gets the same information as plain periodic lines plus one +line per archived file, with no cursor movement — a captured log stays readable. + `CHAT` accepts a numeric id as printed by `tdl chat ls`, a username with or without `@`, or a `t.me`/`tg://` link. A Bot API `-100…` id is converted automatically. A link to a single *message* is refused — it names a message, not diff --git a/cmd/tgexport/sync.go b/cmd/tgexport/sync.go index 6b3a6a1..23608a5 100644 --- a/cmd/tgexport/sync.go +++ b/cmd/tgexport/sync.go @@ -142,8 +142,7 @@ func syncCmd(ctx context.Context, args []string) error { } warnCollisions(idx) before := verify.Check(items, idx) - fmt.Fprintf(os.Stderr, "%d media messages, %d already archived, %d to fetch\n", - before.Expected, before.Present, len(before.Todo())) + report.Survey(os.Stderr, before) todo := selectTodo(items, before, *limitItems) if len(todo) == 0 { @@ -155,13 +154,24 @@ func syncCmd(ctx context.Context, args []string) error { return err } - var todoBytes int64 + var todoBytes, largest int64 for _, it := range todo { todoBytes += it.Size() + largest = max(largest, it.Size()) } - fmt.Fprintf(os.Stderr, "fetching %d file(s), %.1f GiB\n", len(todo), float64(todoBytes)/(1<<30)) + report.Plan(os.Stderr, report.PlanInfo{ + Files: len(todo), + Bytes: todoBytes, + Largest: largest, + Budget: budget, + Staging: *staging, + Threads: *threads, + Downloads: *limit, + Uploads: *uploads, + Destination: dst.String(), + }) - rep := report.New(os.Stderr, len(todo), todoBytes) + rep := report.Events(os.Stderr, len(todo), todoBytes) var res pipeline.Result res, runErr = pipeline.Run(ctx, sliceSeq(todo), pipeline.Options{ Pool: pool, @@ -173,7 +183,7 @@ func syncCmd(ctx context.Context, args []string) error { Budget: budget, Confirm: *confirm, Takeout: *takeout, - Report: rep.Update, + Events: rep, // Re-checked during the run, not only before it: an archive of this // size runs for hours, and the destination can fill in the middle. FreeBytes: func(ctx context.Context) (int64, bool) { diff --git a/cmd/uidemo/main.go b/cmd/uidemo/main.go new file mode 100644 index 0000000..10e4b7f --- /dev/null +++ b/cmd/uidemo/main.go @@ -0,0 +1,75 @@ +// Command uidemo renders the sync progress UI with fake data, so the layout can +// be checked without a Telegram session or a multi-hour transfer. +package main + +import ( + "fmt" + "os" + "time" + + "github.com/iyear/tdl/core/tmedia" + + "github.com/tiennm99dev/telegram-exporter/internal/pipeline" + "github.com/tiennm99dev/telegram-exporter/internal/report" + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" + "github.com/tiennm99dev/telegram-exporter/internal/verify" +) + +func item(id int, name string, size int64) tgsource.Item { + return tgsource.Item{MessageID: id, Name: name, Media: &tmedia.Media{Size: size}} +} + +func main() { + files := []tgsource.Item{ + item(4242, "1234567890_4242_1000000000000000001.mp4", 96<<20), + item(4243, "1234567890_4243_1000000000000000002.mp4", 1900<<20), + item(4244, "1234567890_4244_short.jpg", 2<<20), + } + var total int64 + for _, f := range files { + total += f.Size() + } + fmt.Fprintln(os.Stderr, "PikPak root 'mychannel' has 1438 GiB free") + fmt.Fprintln(os.Stderr, "reading mychannel") + fmt.Fprintln(os.Stderr, " 12,000 messages read in 4m31s") + fmt.Fprintln(os.Stderr, "indexing PikPak root 'mychannel'") + fmt.Fprintln(os.Stderr, " 11,406 objects listed in 2m10s") + + survey := verify.Report{Expected: 12000, Present: 11400, Bytes: 518 << 30} + for i := range 2607 { + survey.Absent = append(survey.Absent, 4242+i) + } + for i := range 6 { + survey.Mismatched = append(survey.Mismatched, + verify.Mismatch{MessageID: 14290 + i, Want: 2 << 30, Got: 221 << 20}) + } + report.Survey(os.Stderr, survey) + report.Plan(os.Stderr, report.PlanInfo{ + Files: 2613, Bytes: 79 << 30, Largest: 2 << 30, Budget: 40 << 30, + Staging: "./staging", Threads: 4, Downloads: 2, Uploads: 2, + Destination: "PikPak root 'mychannel'", + }) + + ev := report.Events(os.Stderr, 2613, 79<<30) + st := pipeline.Stats{} + for _, f := range files { + ev.DownloadStart(f) + } + for step := range 30 { + for _, f := range files { + ev.DownloadBytes(f, f.Size()*int64(step+1)/30) + } + st.BytesDone += total / 30 + ev.Stats(st) + time.Sleep(60 * time.Millisecond) + } + for i, f := range files { + ev.DownloadDone(f, nil) + ev.UploadStart(f) + st.Done = i + 1 + ev.Stats(st) + time.Sleep(400 * time.Millisecond) + ev.UploadDone(f, nil) + } + ev.Finish(st) +} diff --git a/go.mod b/go.mod index ee2e563..8fbdd96 100644 --- a/go.mod +++ b/go.mod @@ -32,8 +32,10 @@ require ( github.com/ProtonMail/go-srp v0.0.7 // indirect github.com/ProtonMail/gopenpgp/v3 v3.4.1 // indirect github.com/PuerkitoBio/goquery v1.12.0 // indirect + github.com/VividCortex/ewma v1.2.0 // indirect github.com/a1ex3/zstd-seekable-format-go/pkg v0.10.0 // indirect github.com/abbot/go-http-auth v0.4.0 // indirect + github.com/acarl005/stripansi v0.0.0-20180116102854-5a71ef0e047d // indirect github.com/adrg/xdg v0.5.3 // indirect github.com/anchore/go-lzo v0.1.1 // indirect github.com/andybalholm/brotli v1.2.2 // indirect @@ -157,7 +159,7 @@ require ( github.com/mailru/easyjson v0.9.2 // indirect github.com/mattn/go-colorable v0.1.15 // indirect github.com/mattn/go-isatty v0.0.23 // indirect - github.com/mattn/go-runewidth v0.0.24 // indirect + github.com/mattn/go-runewidth v0.0.28 // indirect github.com/mitchellh/go-homedir v1.1.0 // indirect github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 // indirect github.com/ncw/swift/v2 v2.0.5 // indirect @@ -203,6 +205,8 @@ require ( github.com/tyler-smith/go-bip39 v1.1.0 // indirect github.com/ulikunitz/xz v0.5.15 // indirect github.com/unknwon/goconfig v1.0.0 // indirect + github.com/vbauerster/cupwriter v0.0.4 // indirect + github.com/vbauerster/mpb/v8 v8.16.1 // indirect github.com/wk8/go-ordered-map/v2 v2.1.8 // indirect github.com/xanzy/ssh-agent v0.3.3 // indirect github.com/youmark/pkcs8 v0.0.0-20240726163527-a2c0da244d78 // indirect diff --git a/go.sum b/go.sum index 9a73884..af02a09 100644 --- a/go.sum +++ b/go.sum @@ -51,12 +51,16 @@ github.com/ProtonMail/gopenpgp/v3 v3.4.1 h1:K7uUhSHSJxORZ+RuHpilTT6S4MA2whCRlXNw github.com/ProtonMail/gopenpgp/v3 v3.4.1/go.mod h1:bGdV9f6edhmd581wzXsQCTKdH8bXBbyhkgDKPjwPc6U= github.com/PuerkitoBio/goquery v1.12.0 h1:pAcL4g3WRXekcB9AU/y1mbKez2dbY2AajVhtkO8RIBo= github.com/PuerkitoBio/goquery v1.12.0/go.mod h1:802ej+gV2y7bbIhOIoPY5sT183ZW0YFofScC4q/hIpQ= +github.com/VividCortex/ewma v1.2.0 h1:f58SaIzcDXrSy3kWaHNvuJgJ3Nmz59Zji6XoJR/q1ow= +github.com/VividCortex/ewma v1.2.0/go.mod h1:nz4BbCtbLyFDeC9SUHbtcT5644juEuWfUAUnGx7j5l4= github.com/a1ex3/zstd-seekable-format-go/pkg v0.10.0 h1:iLDOF0rdGTrol/q8OfPIIs5kLD8XvA2q75o6Uq/tgak= github.com/a1ex3/zstd-seekable-format-go/pkg v0.10.0/go.mod h1:DrEWcQJjz7t5iF2duaiyhg4jyoF0kxOD6LtECNGkZ/Q= github.com/aalpar/deheap v1.1.2 h1:MABHLcnjqsffb8GLkUFDigqpBBxOMz0DoKM9QfELeTw= github.com/aalpar/deheap v1.1.2/go.mod h1:A+nfkD4JbS05sewV0he/MYgR/90vfqyMoNNROgs+rmA= github.com/abbot/go-http-auth v0.4.0 h1:QjmvZ5gSC7jm3Zg54DqWE/T5m1t2AfDu6QlXJT0EVT0= github.com/abbot/go-http-auth v0.4.0/go.mod h1:Cz6ARTIzApMJDzh5bRMSUou6UMSp0IEXg9km/ci7TJM= +github.com/acarl005/stripansi v0.0.0-20180116102854-5a71ef0e047d h1:licZJFw2RwpHMqeKTCYkitsPqHNxTmd4SNR5r94FGM8= +github.com/acarl005/stripansi v0.0.0-20180116102854-5a71ef0e047d/go.mod h1:asat636LX7Bqt5lYEZ27JNDcqxfjdBQuJ/MM4CN/Lzo= github.com/adrg/xdg v0.5.3 h1:xRnxJXne7+oWDatRhR1JLnvuccuIeCoBu2rtuLqQB78= github.com/adrg/xdg v0.5.3/go.mod h1:nlTsY+NNiCBGCK2tpm09vRqfVzrc2fLmXGpBLF0zlTQ= github.com/anchore/go-lzo v0.1.1 h1:IwL/fvkdtlIrYIXck6WxZ3nb8WjjHziYYmGxlooyOnM= @@ -384,6 +388,8 @@ github.com/mattn/go-isatty v0.0.23/go.mod h1:nMCL3Zebbrt45jsMDgnfIwz6ydEQApk5oEI github.com/mattn/go-runewidth v0.0.3/go.mod h1:LwmH8dsx7+W8Uxz3IHJYH5QSwggIsqBzpuz5H//U1FU= github.com/mattn/go-runewidth v0.0.24 h1:cpokDiIn0MGnhdHwuWnJBITySJ20QyNGnY2kR/ay2DU= github.com/mattn/go-runewidth v0.0.24/go.mod h1:XBkDxAl56ILZc9knddidhrOlY5R/pDhgLpndooCuJAs= +github.com/mattn/go-runewidth v0.0.28 h1:rPyg2ybwEKPebvpzVWe1gKBkH8EQFkxO4Y0hjBeLaBU= +github.com/mattn/go-runewidth v0.0.28/go.mod h1:3qAiGCV4Koz/yuveO58qUefmUTRm8r0IGEXZ9jeHp/8= github.com/mitchellh/go-homedir v1.1.0 h1:lukF9ziXFxDFPkA1vsr5zpc1XuPDn/wFntq5mG+4E0Y= github.com/mitchellh/go-homedir v1.1.0/go.mod h1:SfyaCUpYCn1Vlf4IUYiD9fPX4A5wJrkLzIz1N1q0pr0= github.com/moby/sys/mountinfo v0.7.2 h1:1shs6aH5s4o5H2zQLn796ADW1wMrIwHsyJ2v9KouLrg= @@ -526,6 +532,10 @@ github.com/ulikunitz/xz v0.5.15 h1:9DNdB5s+SgV3bQ2ApL10xRc35ck0DuIX/isZvIk+ubY= github.com/ulikunitz/xz v0.5.15/go.mod h1:nbz6k7qbPmH4IRqmfOplQw/tblSgqTqBwxkY0oWt/14= github.com/unknwon/goconfig v1.0.0 h1:rS7O+CmUdli1T+oDm7fYj1MwqNWtEJfNj+FqcUHML8U= github.com/unknwon/goconfig v1.0.0/go.mod h1:qu2ZQ/wcC/if2u32263HTVC39PeOQRSmidQk3DuDFQ8= +github.com/vbauerster/cupwriter v0.0.4 h1:9sBPe0uXWLZuWQU5lqVbhyFlxX6c09asST/YfatFAys= +github.com/vbauerster/cupwriter v0.0.4/go.mod h1:IFyzS6Xis5dnBH/rdAhrnuzg3c+KkUqEN6yE8lhJlDw= +github.com/vbauerster/mpb/v8 v8.16.1 h1:gNYmwMip9xRWNGAiblZOgUNXWeU2P0NIGd5x0f8ffbc= +github.com/vbauerster/mpb/v8 v8.16.1/go.mod h1:gnU8zNF/JWltFepqwko/ulMEUIDrydIq7T4UdMN26Nw= github.com/wk8/go-ordered-map/v2 v2.1.8 h1:5h/BUHu93oj4gIdvHHHGsScSTMijfx5PeYkE/fJgbpc= github.com/wk8/go-ordered-map/v2 v2.1.8/go.mod h1:5nJHM5DyteebpVlHnWMV0rPz6Zp7+xBAnxjb1X5vnTw= github.com/xanzy/ssh-agent v0.3.3 h1:+/15pJfg/RsTxqYcX6fHqOXZwwMP+2VyYWJeWM2qQFM= diff --git a/internal/pipeline/download.go b/internal/pipeline/download.go index 7d9bf8a..daa3718 100644 --- a/internal/pipeline/download.go +++ b/internal/pipeline/download.go @@ -24,10 +24,10 @@ type DownloadOptions struct { Limit int // files in flight Takeout bool // use a takeout session, as `tdl dl --takeout` does - // Report, when set, is called with a running summary. It is invoked from - // download worker goroutines, so it must be cheap and safe to call + // Events, when set, receives the run's per-item lifecycle. It is invoked + // from download worker goroutines, so it must be cheap and safe to call // concurrently. - Report func(Stats) + Events Events // acquire reserves staging space before a download starts, blocking until // there is room. Unset means no bound. release hands a reservation back for @@ -87,7 +87,7 @@ func Download(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Downlo o.onFailed(e.item) } return ferr - }, o.Report) + }, o.Events) err := downloader.New(downloader.Options{ Pool: o.Pool, diff --git a/internal/pipeline/events.go b/internal/pipeline/events.go new file mode 100644 index 0000000..8356ce6 --- /dev/null +++ b/internal/pipeline/events.go @@ -0,0 +1,38 @@ +package pipeline + +import "github.com/tiennm99dev/telegram-exporter/internal/tgsource" + +// Events receives a run's per-item lifecycle. +// +// Aggregate Stats alone cannot say what a run is doing right now — which files +// are moving, how far along each is, whether the time is going into downloads or +// uploads. On an archive that runs for hours those are the questions being +// asked, and a single "26/2613 done" line answers none of them. +// +// Every method is called from a worker goroutine, several at once, and some of +// them fire per network chunk. Implementations must be safe to call +// concurrently and must not block: a slow renderer would throttle the transfers +// it is describing. +type Events interface { + // Stats reports the running totals. + Stats(Stats) + + DownloadStart(it tgsource.Item) + // DownloadBytes reports the total written for it so far, not a delta. + DownloadBytes(it tgsource.Item, done int64) + DownloadDone(it tgsource.Item, err error) + + UploadStart(it tgsource.Item) + UploadDone(it tgsource.Item, err error) +} + +// nopEvents is used when a caller wants no reporting, so nothing on the hot +// path has to nil-check. +type nopEvents struct{} + +func (nopEvents) Stats(Stats) {} +func (nopEvents) DownloadStart(tgsource.Item) {} +func (nopEvents) DownloadBytes(tgsource.Item, int64) {} +func (nopEvents) DownloadDone(tgsource.Item, error) {} +func (nopEvents) UploadStart(tgsource.Item) {} +func (nopEvents) UploadDone(tgsource.Item, error) {} diff --git a/internal/pipeline/pipeline.go b/internal/pipeline/pipeline.go index a0d6cc7..def6a24 100644 --- a/internal/pipeline/pipeline.go +++ b/internal/pipeline/pipeline.go @@ -42,7 +42,9 @@ type Options struct { FreeBytes func(context.Context) (int64, bool) MinFree int64 - Report func(Stats) + // Events, when set, receives the run's per-item lifecycle: which files are + // downloading, which are uploading, and how far along each one is. + Events Events } // Result is what a run achieved. @@ -108,6 +110,9 @@ func Run(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Options) (R if o.MaxFailures <= 0 { o.MaxFailures = 5 } + if o.Events == nil { + o.Events = nopEvents{} + } local, err := fs.NewFs(ctx, o.Staging) if err != nil { @@ -145,7 +150,9 @@ func Run(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Options) (R // attempt, so the file is dropped rather than uploaded. err := guard.check(ctx) if err == nil { + o.Events.UploadStart(it) err = up.upload(ctx, it) + o.Events.UploadDone(it, err) } if err != nil { @@ -186,7 +193,7 @@ func Run(ctx context.Context, seq iter.Seq2[tgsource.Item, error], o Options) (R Threads: o.Threads, Limit: o.Limit, Takeout: o.Takeout, - Report: o.Report, + Events: o.Events, acquire: budget.acquire, release: budget.release, onReady: func(it tgsource.Item) { uploads <- it }, diff --git a/internal/pipeline/progress.go b/internal/pipeline/progress.go index 8c7c75c..5ad97da 100644 --- a/internal/pipeline/progress.go +++ b/internal/pipeline/progress.go @@ -36,14 +36,17 @@ type progress struct { inFlight map[int]int64 // message id -> bytes written so far finish func(*elem, error) error - report func(Stats) + events Events } -func newProgress(finish func(*elem, error) error, report func(Stats)) *progress { +func newProgress(finish func(*elem, error) error, events Events) *progress { + if events == nil { + events = nopEvents{} + } return &progress{ inFlight: make(map[int]int64), finish: finish, - report: report, + events: events, } } @@ -54,7 +57,8 @@ func (p *progress) OnAdd(e downloader.Elem) { p.stats.BytesTotal += el.item.Size() stats := p.stats p.mu.Unlock() - p.emit(stats) + p.events.DownloadStart(el.item) + p.events.Stats(stats) } func (p *progress) OnDownload(e downloader.Elem, state downloader.ProgressState) { @@ -67,7 +71,8 @@ func (p *progress) OnDownload(e downloader.Elem, state downloader.ProgressState) p.stats.BytesDone += state.Downloaded - prev stats := p.stats p.mu.Unlock() - p.emit(stats) + p.events.DownloadBytes(el.item, state.Downloaded) + p.events.Stats(stats) } func (p *progress) OnDone(e downloader.Elem, err error) { @@ -89,13 +94,8 @@ func (p *progress) OnDone(e downloader.Elem, err error) { p.outcomes = append(p.outcomes, Outcome{Item: el.item, Err: err}) stats := p.stats p.mu.Unlock() - p.emit(stats) -} - -func (p *progress) emit(s Stats) { - if p.report != nil { - p.report(s) - } + p.events.DownloadDone(el.item, err) + p.events.Stats(stats) } func (p *progress) results() ([]Outcome, Stats) { diff --git a/internal/report/live.go b/internal/report/live.go new file mode 100644 index 0000000..d8410a6 --- /dev/null +++ b/internal/report/live.go @@ -0,0 +1,219 @@ +package report + +import ( + "fmt" + "io" + "sync" + "time" + + "github.com/vbauerster/mpb/v8" + "github.com/vbauerster/mpb/v8/decor" + + "github.com/tiennm99dev/telegram-exporter/internal/pipeline" + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" +) + +// nameWidth is how much of a filename a per-file bar shows. Telegram names run +// to 100+ characters and the bar has to fit beside them. +const nameWidth = 34 + +// Live renders a run as a set of progress bars: one overall, plus one for each +// file currently moving. +// +// The aggregate line it replaces could say how much was done but never what was +// happening — which files were in flight, whether a stall was a slow download or +// a slow upload, how long the rest would take. On a run measured in hours those +// are the only questions worth answering. +// +// Bars are for terminals only. A redirected run gets lineEvents instead, because +// this writes ANSI cursor movement continuously and a captured log of it is +// unreadable. +type Live struct { + w io.Writer + p *mpb.Progress + total *mpb.Bar + + mu sync.Mutex + down map[int]*mpb.Bar + up map[int]*mpb.Bar + // done is read by the overall bar's decorator from mpb's render goroutine, + // so it is guarded by the same lock as the maps. + done int + + // seq gives each per-file bar a distinct, increasing priority so bars keep + // their position between frames. Sharing one priority lets mpb reorder them + // on every redraw, which makes a steady transfer look like it is thrashing. + seq int + files int + start time.Time +} + +// NewLive builds a bar renderer over w for a run of the given size. +func NewLive(w io.Writer, files int, bytes int64) *Live { + p := mpb.New( + mpb.WithOutput(w), + mpb.WithWidth(28), + mpb.WithRefreshRate(120*time.Millisecond), + ) + l := &Live{ + w: w, + p: p, + down: make(map[int]*mpb.Bar), + up: make(map[int]*mpb.Bar), + files: files, + start: time.Now(), + } + l.total = p.New(bytes, + mpb.BarStyle().Lbound("[").Filler("=").Tip(">").Padding(" ").Rbound("]"), + mpb.BarPriority(0), + mpb.BarNoPop(), + mpb.PrependDecorators( + decor.Name(" total ", decor.WC{W: 9}), + decor.Any(func(decor.Statistics) string { return l.counts() }, decor.WC{W: 16}), + ), + mpb.AppendDecorators( + decor.CountersKibiByte("% .1f / % .1f", decor.WC{W: 20}), + decor.AverageSpeed(decor.SizeB1024(0), " % .1f", decor.WC{W: 12}), + decor.OnComplete(decor.AverageETA(decor.ET_STYLE_GO, decor.WC{W: 10}), ""), + ), + ) + return l +} + +// Bars are grouped by band: the total on top, then downloads, then uploads. +// Within a band they are ordered by when they started. +const ( + downloadBand = 1 << 20 + uploadBand = 1 << 21 +) + +// next allocates the priority for a new bar in the given band. +func (l *Live) next(band int) int { + l.mu.Lock() + defer l.mu.Unlock() + l.seq++ + return band + l.seq +} + +func (l *Live) counts() string { + l.mu.Lock() + defer l.mu.Unlock() + return fmt.Sprintf("%s/%s files", humanCount(l.done), humanCount(l.files)) +} + +// Stats advances the overall bar. +func (l *Live) Stats(s pipeline.Stats) { + l.mu.Lock() + l.done = s.Done + s.Failed + l.mu.Unlock() + // The bar's own counters, speed and ETA all derive from this, so it is what + // makes the totals move rather than sitting at zero. + l.total.SetCurrent(s.BytesDone) +} + +func (l *Live) DownloadStart(it tgsource.Item) { + bar := l.p.New(it.Size(), + mpb.BarStyle().Lbound("[").Filler("=").Tip(">").Padding(" ").Rbound("]"), + mpb.BarRemoveOnComplete(), + mpb.BarPriority(l.next(downloadBand)), + mpb.PrependDecorators( + decor.Name(" ↓ "), + decor.Name(short(it.Name), decor.WC{W: nameWidth + 2, C: decor.DindentRight}), + ), + mpb.AppendDecorators( + decor.CountersKibiByte("% .1f / % .1f", decor.WC{W: 20}), + decor.AverageSpeed(decor.SizeB1024(0), " % .1f", decor.WC{W: 12}), + ), + ) + l.mu.Lock() + l.down[it.MessageID] = bar + l.mu.Unlock() +} + +func (l *Live) DownloadBytes(it tgsource.Item, done int64) { + l.mu.Lock() + bar := l.down[it.MessageID] + l.mu.Unlock() + if bar != nil { + bar.SetCurrent(done) + } +} + +func (l *Live) DownloadDone(it tgsource.Item, err error) { + l.mu.Lock() + bar := l.down[it.MessageID] + delete(l.down, it.MessageID) + l.mu.Unlock() + if bar == nil { + return + } + // Aborted rather than completed on failure, so a bar for a file that never + // arrived does not linger at 100%. + if err != nil { + bar.Abort(true) + return + } + bar.SetCurrent(it.Size()) +} + +func (l *Live) UploadStart(it tgsource.Item) { + // No byte-level callbacks exist for the upload leg — rclone's MoveFile is a + // single blocking call — so this is a spinner, not a bar. Showing which file + // is uploading is the point: a run that looks stalled is usually waiting on + // one large object, and until now nothing said so. + bar := l.p.New(0, mpb.SpinnerStyle().PositionLeft(), + mpb.BarRemoveOnComplete(), + mpb.BarPriority(l.next(uploadBand)), + mpb.PrependDecorators( + decor.Name(" ↑ "), + decor.Name(short(it.Name), decor.WC{W: nameWidth + 2, C: decor.DindentRight}), + ), + mpb.AppendDecorators( + decor.Any(func(decor.Statistics) string { + return fmt.Sprintf("uploading %s", humanBytes(it.Size())) + }, decor.WC{W: 24}), + ), + ) + l.mu.Lock() + l.up[it.MessageID] = bar + l.mu.Unlock() +} + +func (l *Live) UploadDone(it tgsource.Item, _ error) { + l.mu.Lock() + bar := l.up[it.MessageID] + delete(l.up, it.MessageID) + l.mu.Unlock() + if bar != nil { + bar.Abort(true) + } +} + +// Finish drains the bars and prints the closing summary. +func (l *Live) Finish(s pipeline.Stats) { + l.mu.Lock() + for _, b := range l.down { + b.Abort(true) + } + for _, b := range l.up { + b.Abort(true) + } + clear(l.down) + clear(l.up) + l.mu.Unlock() + + l.total.Abort(true) + l.p.Wait() + writeSummary(l.w, s, time.Since(l.start)) +} + +// short trims a filename to fit beside a bar, keeping the end — the extension +// and the distinguishing digits are there, while the shared dialog-id prefix is +// not. +func short(name string) string { + r := []rune(name) + if len(r) <= nameWidth { + return name + } + return "…" + string(r[len(r)-nameWidth+1:]) +} diff --git a/internal/report/progress.go b/internal/report/progress.go index 4d31ec9..4e73b43 100644 --- a/internal/report/progress.go +++ b/internal/report/progress.go @@ -9,6 +9,7 @@ import ( "time" "github.com/tiennm99dev/telegram-exporter/internal/pipeline" + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" ) // statsInterval is how often a redirected run prints a line. @@ -19,10 +20,14 @@ import ( // else gets a periodic summary. const statsInterval = 30 * time.Second -// Reporter renders progress, adapting to whether it is writing to a terminal. +// Reporter renders progress as periodic plain-text lines. +// +// This is the redirected-output path. Bars are deliberately absent: they are +// continuous ANSI cursor movement, and a captured log of them is megabytes of +// control characters — which is exactly what tdl's progress bar did to the shell +// pipeline's logs. type Reporter struct { w io.Writer - tty bool total int totalBytes int64 @@ -31,35 +36,40 @@ type Reporter struct { lastLine time.Time } -// New builds a reporter for w, which is treated as a terminal when it is one. +// Events builds the renderer suited to w: bars on a terminal, periodic lines +// anywhere else. // // The totals are passed in rather than taken from Stats because Stats.BytesTotal // only counts items the downloader has started, so a progress line built from it // shows a denominator that grows as the run proceeds — "0 B of 52 KiB" on a run // that will move gigabytes. The caller knows the real figures before starting. -func New(w io.Writer, total int, totalBytes int64) *Reporter { - return &Reporter{w: w, tty: isTerminal(w), total: total, totalBytes: totalBytes, started: time.Now()} +func Events(w io.Writer, total int, totalBytes int64) interface { + pipeline.Events + Finish(pipeline.Stats) +} { + if isTerminal(w) { + return NewLive(w, total, totalBytes) + } + return newReporter(w, total, totalBytes) } -// Update renders a snapshot. Safe to call from several goroutines. +func newReporter(w io.Writer, total int, totalBytes int64) *Reporter { + return &Reporter{w: w, total: total, totalBytes: totalBytes, started: time.Now()} +} + +// Stats renders a snapshot. Safe to call from several goroutines. // // A contended update is dropped rather than queued. Every download worker calls // this on each progress callback, so holding the lock across the write would -// make terminal latency — an ssh session with a slow link, say — throttle the -// downloads themselves. A skipped frame costs nothing; the next callback is -// milliseconds away and Finish always prints. -func (r *Reporter) Update(s pipeline.Stats) { +// make write latency throttle the downloads themselves. A skipped frame costs +// nothing; the next callback is milliseconds away and Finish always prints. +func (r *Reporter) Stats(s pipeline.Stats) { if !r.mu.TryLock() { return } defer r.mu.Unlock() now := time.Now() - if r.tty { - // \r rather than \n: one line, redrawn. - fmt.Fprintf(r.w, "\r\033[K%s", r.line(s, now)) - return - } if now.Sub(r.lastLine) < statsInterval { return } @@ -67,25 +77,49 @@ func (r *Reporter) Update(s pipeline.Stats) { fmt.Fprintf(r.w, "%s\n", r.line(s, now)) } -// Finish writes the closing summary, ending the redrawn line if there was one. +// The per-file events are recorded as one line each rather than a bar. At one +// line per file this stays readable in a log, and it is what makes a captured +// run auditable afterwards: which files moved, in what order, and which failed. +func (r *Reporter) DownloadStart(tgsource.Item) {} +func (r *Reporter) DownloadBytes(tgsource.Item, int64) {} +func (r *Reporter) UploadStart(tgsource.Item) {} + +func (r *Reporter) DownloadDone(it tgsource.Item, err error) { + if err != nil { + r.mu.Lock() + defer r.mu.Unlock() + fmt.Fprintf(r.w, " download failed %q: %v\n", it.Name, err) + } +} + +func (r *Reporter) UploadDone(it tgsource.Item, err error) { + r.mu.Lock() + defer r.mu.Unlock() + if err != nil { + fmt.Fprintf(r.w, " upload failed %q: %v\n", it.Name, err) + return + } + fmt.Fprintf(r.w, " archived %-10s %q\n", humanBytes(it.Size()), it.Name) +} + +// Finish writes the closing summary. func (r *Reporter) Finish(s pipeline.Stats) { r.mu.Lock() defer r.mu.Unlock() - if r.tty { - fmt.Fprint(r.w, "\r\033[K") - } - elapsed := time.Since(r.started).Round(time.Second) - fmt.Fprintf(r.w, "%d done, %d failed, %s in %s (%s/s)\n", - s.Done, s.Failed, humanBytes(s.BytesDone), elapsed, - humanBytes(int64(float64(s.BytesDone)/max(elapsed.Seconds(), 1)))) + writeSummary(r.w, s, time.Since(r.started)) } func (r *Reporter) line(s pipeline.Stats, now time.Time) string { elapsed := now.Sub(r.started) rate := float64(s.BytesDone) / max(elapsed.Seconds(), 1) - return fmt.Sprintf("%d/%d done, %d failed, %s of %s, %s/s", - s.Done, r.total, s.Failed, - humanBytes(s.BytesDone), humanBytes(r.totalBytes), humanBytes(int64(rate))) + eta := "—" + if rate > 0 && r.totalBytes > s.BytesDone { + eta = time.Duration(float64(r.totalBytes-s.BytesDone) / rate * float64(time.Second)). + Round(time.Second).String() + } + return fmt.Sprintf(" %s/%s files, %d failed, %s of %s, %s/s, ETA %s", + humanCount(s.Done), humanCount(r.total), s.Failed, + humanBytes(s.BytesDone), humanBytes(r.totalBytes), humanBytes(int64(rate)), eta) } func humanBytes(n int64) string { @@ -117,3 +151,16 @@ func isTerminal(w io.Writer) bool { } return info.Mode()&os.ModeCharDevice != 0 } + +// writeSummary prints the closing line both renderers end with. +func writeSummary(w io.Writer, s pipeline.Stats, elapsed time.Duration) { + elapsed = elapsed.Round(time.Second) + fmt.Fprintf(w, "%s done, %s failed, %s in %s (%s/s)\n", + humanCount(s.Done), humanCount(s.Failed), humanBytes(s.BytesDone), elapsed, + humanBytes(int64(float64(s.BytesDone)/max(elapsed.Seconds(), 1)))) +} + +// HumanBytes and HumanCount are the shared formatters, exported so the commands +// print the same shapes as the progress renderers do. +func HumanBytes(n int64) string { return humanBytes(n) } +func HumanCount(n int) string { return humanCount(n) } diff --git a/internal/report/summary.go b/internal/report/summary.go new file mode 100644 index 0000000..9b8ae45 --- /dev/null +++ b/internal/report/summary.go @@ -0,0 +1,61 @@ +package report + +import ( + "fmt" + "io" + + "github.com/tiennm99dev/telegram-exporter/internal/verify" +) + +// writeSurvey states what the chat holds and how much of it is already archived, +// broken out by why each outstanding file is outstanding. The single "N to +// fetch" it replaces hid the difference between never-fetched, empty, +// wrong-size, and unarchivable — which is the difference between a run that will +// converge and one that cannot. +func Survey(w io.Writer, r verify.Report) { + fmt.Fprintf(w, "\n chat holds %s media, %s\n", + humanCount(r.Expected), humanBytes(r.Bytes)) + fmt.Fprintf(w, " archived %s\n", humanCount(r.Present)) + + for _, row := range []struct { + label string + n int + }{ + {"never fetched", len(r.Absent) - len(r.Unsafe)}, + {"zero-byte", len(r.ZeroByte)}, + {"wrong size", len(r.Mismatched)}, + {"unarchivable", len(r.Unsafe)}, + } { + if row.n > 0 { + fmt.Fprintf(w, " %-13s %s\n", row.label, humanCount(row.n)) + } + } +} + +// PlanInfo is what the run is about to do, stated before it starts. +type PlanInfo struct { + Files int + Bytes int64 + Largest int64 + Budget int64 + Staging string + Threads int + Downloads, Uploads int + Destination string +} + +// writePlan prints the settings that decide how long the run takes and how much +// disk it uses, so an operator can stop it before a multi-hour transfer rather +// than discover the wrong cap partway through. +func Plan(w io.Writer, p PlanInfo) { + cap := "uncapped" + if p.Budget > 0 { + cap = humanBytes(p.Budget) + } + fmt.Fprintf(w, "\n fetching %s files, %s (largest %s)\n", + humanCount(p.Files), humanBytes(p.Bytes), humanBytes(p.Largest)) + fmt.Fprintf(w, " into %s\n", p.Destination) + fmt.Fprintf(w, " staging %s, capped at %s\n", p.Staging, cap) + fmt.Fprintf(w, " concurrency %d download(s) x %d thread(s), %d upload(s)\n\n", + p.Downloads, p.Threads, p.Uploads) +} diff --git a/internal/report/summary_test.go b/internal/report/summary_test.go new file mode 100644 index 0000000..692b6fd --- /dev/null +++ b/internal/report/summary_test.go @@ -0,0 +1,95 @@ +package report + +import ( + "strings" + "testing" + + "github.com/iyear/tdl/core/tmedia" + + "github.com/tiennm99dev/telegram-exporter/internal/pipeline" + "github.com/tiennm99dev/telegram-exporter/internal/tgsource" + "github.com/tiennm99dev/telegram-exporter/internal/verify" +) + +func TestSurveyBreaksOutdWhyFilesAreOutstanding(t *testing.T) { + r := verify.Report{Expected: 12000, Present: 11400, Bytes: 518 << 30} + r.Absent = append(r.Absent, 1, 2, 3) + r.ZeroByte = append(r.ZeroByte, 4) + r.Mismatched = append(r.Mismatched, verify.Mismatch{MessageID: 5}) + r.Unsafe = append(r.Unsafe, verify.Unsafe{MessageID: 3}) + + var sb strings.Builder + Survey(&sb, r) + out := sb.String() + + for _, want := range []string{"12,000 media", "11,400", "zero-byte", "wrong size", "unarchivable"} { + if !strings.Contains(out, want) { + t.Errorf("survey missing %q:\n%s", want, out) + } + } + // Unsafe ids are inside Absent, so counting both would double-count them and + // the rows would not add up to the outstanding total. + if !strings.Contains(out, "never fetched 2") { + t.Errorf("never-fetched should exclude the unarchivable id:\n%s", out) + } +} + +func TestSurveyOmitsEmptyRows(t *testing.T) { + var sb strings.Builder + Survey(&sb, verify.Report{Expected: 10, Present: 10}) + if strings.Contains(sb.String(), "zero-byte") { + t.Errorf("a clean archive should list no failure rows:\n%s", sb.String()) + } +} + +func TestPlanStatesTheCap(t *testing.T) { + var sb strings.Builder + Plan(&sb, PlanInfo{Files: 2613, Bytes: 79 << 30, Largest: 2 << 30, + Staging: "./staging", Threads: 4, Downloads: 2, Uploads: 2, Destination: "remote:x"}) + if !strings.Contains(sb.String(), "uncapped") { + t.Errorf("a run with no budget must say so:\n%s", sb.String()) + } + + sb.Reset() + Plan(&sb, PlanInfo{Files: 1, Budget: 40 << 30, Staging: "./staging", Destination: "remote:x"}) + if !strings.Contains(sb.String(), "40.0 GiB") { + t.Errorf("plan should state the cap:\n%s", sb.String()) + } +} + +// The redirected path must stay free of cursor movement: bars in a captured log +// are megabytes of control characters, which is what tdl's progress bar did to +// the shell pipeline's logs. +func TestReporterWritesNoAnsi(t *testing.T) { + var sb strings.Builder + r := newReporter(&sb, 3, 300) + it := tgsource.Item{MessageID: 1, Name: "a.mp4", Media: &tmedia.Media{Size: 100}} + + r.DownloadStart(it) + for i := range 500 { + r.DownloadBytes(it, int64(i)) + r.Stats(pipeline.Stats{Done: 1, BytesDone: int64(i)}) + } + r.DownloadDone(it, nil) + r.UploadStart(it) + r.UploadDone(it, nil) + r.Finish(pipeline.Stats{Done: 1, BytesDone: 100}) + + out := sb.String() + if strings.ContainsAny(out, "\r\033") { + t.Errorf("redirected output contains control characters: %q", out) + } + // A completed file is still recorded, so a captured run says what moved. + if !strings.Contains(out, "archived") || !strings.Contains(out, "a.mp4") { + t.Errorf("redirected output should record each archived file:\n%s", out) + } +} + +// Events picks the renderer from the writer: a strings.Builder is not a +// terminal, so it must never get bars. +func TestEventsChoosesTheLineRendererOffTerminal(t *testing.T) { + var sb strings.Builder + if _, ok := Events(&sb, 1, 1).(*Reporter); !ok { + t.Error("a non-terminal writer got the bar renderer") + } +} From b05247abed7585edf7f581e69fe8396f183169f2 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Sun, 6 Sep 2026 21:54:38 +0700 Subject: [PATCH 13/16] fix: retry a failed upload instead of abandoning the file MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit pikpak commits an upload as a server-side async task, and rclone polls that task only as long as its low-level retries last. A task still in PHASE_TYPE_PENDING when that budget runs out is reported as "can't verify the task is completed" — and checking the live archive ten minutes later, those tasks had not committed: the objects were simply absent. The transfer itself was fine; only the confirmation timed out. Abandoning the file meant re-downloading it from Telegram on the next pass, which for this archive can be two gigabytes, so a run against a slow remote paid for the same bytes repeatedly. The staged copy is still on disk when a move fails — rclone says so in the same breath — so another attempt costs seconds instead. Every attempt after the first checks the remote before re-uploading. A pending task may have committed during the backoff, and pikpak allows two files under one name, so re-uploading blind is how one file becomes two — which verification then reports as an ambiguous basename on every future run. rclone's log now goes through the renderer. It writes to stderr on its own schedule, so its error lines were landing mid-redraw and shredding the display exactly when there was most to read. Also: "capped at uncapped" now reads "uncapped", and the ETA column fits the three-digit hour counts a slow remote produces. --- cmd/tgexport/sync.go | 5 + cmd/uidemo/main.go | 9 ++ internal/pipeline/upload.go | 130 ++++++++++++++++++------ internal/pipeline/upload_test.go | 169 ++++++++++++++++++++++--------- internal/report/live.go | 12 ++- internal/report/rclonelog.go | 30 ++++++ internal/report/summary.go | 4 +- 7 files changed, 280 insertions(+), 79 deletions(-) create mode 100644 internal/report/rclonelog.go diff --git a/cmd/tgexport/sync.go b/cmd/tgexport/sync.go index 23608a5..5360b10 100644 --- a/cmd/tgexport/sync.go +++ b/cmd/tgexport/sync.go @@ -172,6 +172,11 @@ func syncCmd(ctx context.Context, args []string) error { }) rep := report.Events(os.Stderr, len(todo), todoBytes) + // rclone logs to stderr on its own schedule, which lands in the middle + // of a bar redraw. Routing it through the renderer keeps both readable. + if live, ok := rep.(*report.Live); ok { + defer report.CaptureRcloneLog(ctx, live.LogWriter())() + } var res pipeline.Result res, runErr = pipeline.Run(ctx, sliceSeq(todo), pipeline.Options{ Pool: pool, diff --git a/cmd/uidemo/main.go b/cmd/uidemo/main.go index 10e4b7f..4822acc 100644 --- a/cmd/uidemo/main.go +++ b/cmd/uidemo/main.go @@ -3,11 +3,13 @@ package main import ( + "context" "fmt" "os" "time" "github.com/iyear/tdl/core/tmedia" + "github.com/rclone/rclone/fs" "github.com/tiennm99dev/telegram-exporter/internal/pipeline" "github.com/tiennm99dev/telegram-exporter/internal/report" @@ -51,11 +53,18 @@ func main() { }) ev := report.Events(os.Stderr, 2613, 79<<30) + if live, ok := ev.(*report.Live); ok { + defer report.CaptureRcloneLog(context.Background(), live.LogWriter())() + } st := pipeline.Stats{} for _, f := range files { ev.DownloadStart(f) } for step := range 30 { + if step == 12 { + fs.Errorf(nil, "1234567890_4245_1000000000000000003.jpg: Failed to copy: "+ + "can't verify the task is completed") + } for _, f := range files { ev.DownloadBytes(f, f.Size()*int64(step+1)/30) } diff --git a/internal/pipeline/upload.go b/internal/pipeline/upload.go index 47670eb..29ef56d 100644 --- a/internal/pipeline/upload.go +++ b/internal/pipeline/upload.go @@ -12,6 +12,15 @@ import ( "github.com/tiennm99dev/telegram-exporter/internal/tgsource" ) +// uploadAttempts is how many times a file is offered to the remote before the +// run gives up on it. +// +// Retrying here rather than fetching again next pass is the whole point: when a +// move fails the local copy is still in staging, so another attempt costs a few +// seconds, while abandoning it costs re-downloading the file from Telegram — +// which for this archive can be two gigabytes. +const uploadAttempts = 3 + // uploader moves finished files from staging to the destination remote. type uploader struct { local fs.Fs // the staging directory as an rclone filesystem @@ -19,28 +28,56 @@ type uploader struct { confirm bool } -// upload moves one finished file to the remote and, unless disabled, proves it -// arrived at the expected size. +// upload moves one finished file to the remote, retrying a failure. +// +// pikpak is why this retries at all. It commits an upload as a server-side async +// task, and rclone polls that task only as long as its low-level retries last +// (backend/pikpak/helper.go:205-214, bounded by fs.NewPacer). A task still in +// PHASE_TYPE_PENDING when the polling budget runs out is reported as +// "can't verify the task is completed", and observed against the live archive +// that task then never commits — the object is simply absent afterwards. The +// transfer itself was fine; only the confirmation timed out. +func (u *uploader) upload(ctx context.Context, it tgsource.Item) error { + var errs []error + for attempt := 1; ; attempt++ { + // Every attempt after the first looks before it leaps. A pending task + // from the previous attempt may have committed during the backoff, and + // pikpak allows two files with the same name — so re-uploading without + // checking is how one file becomes two, which verification then reports + // as an ambiguous basename forever. + if attempt > 1 { + landed, err := u.settle(ctx, it) + if err != nil { + errs = append(errs, err) + } + if landed { + return nil + } + } + + err := u.attempt(ctx, it) + if err == nil { + return nil + } + errs = append(errs, fmt.Errorf("attempt %d/%d: %w", attempt, uploadAttempts, err)) + + if attempt >= uploadAttempts || ctx.Err() != nil { + return errors.Join(errs...) + } + if err := sleep(ctx, uploadBackoff(attempt)); err != nil { + return errors.Join(append(errs, err)...) + } + } +} + +// attempt runs one move and, unless disabled, proves the object arrived at the +// expected size. // // MoveFile removes the local copy as part of the move, so a successful return // means the file is on the remote and off local disk. -// -// Both failure paths end at dropShort, and that is the part that matters. An -// object of the wrong size sitting under the right name is worse than no object -// at all: verification matches on name and non-zero size, so it would be counted -// archived by this run and by every run after it — permanently, once the local -// copy is gone. Removing it turns a silent corruption into an absent file the -// next run fetches again. -func (u *uploader) upload(ctx context.Context, it tgsource.Item) error { +func (u *uploader) attempt(ctx context.Context, it tgsource.Item) error { if err := operations.MoveFile(ctx, u.dst, u.local, it.Name, it.Name); err != nil { - err = fmt.Errorf("move %q to %s: %w", it.Name, u.dst.String(), err) - // A failed move can still leave a partial object under the final name. - // rclone only writes to a temporary name when the backend advertises - // PartialUploads (copy.go:93), and it only cleans up after itself when - // it did (copy.go:348-350) — pikpak, the remote this was built against, - // advertises neither, so a died-halfway transfer stays exactly where a - // complete one would be. - return errors.Join(err, u.dropShort(ctx, it)) + return fmt.Errorf("move %q to %s: %w", it.Name, u.dst.String(), err) } if !u.confirm { return nil @@ -61,13 +98,19 @@ func (u *uploader) upload(ctx context.Context, it tgsource.Item) error { return nil } -// dropShort removes an object left under it.Name at the wrong size. +// settle reports whether the file is on the remote at the right size, cleaning +// up if it is there at the wrong one. // -// An object of the *right* size is deliberately left alone. pikpak commits an -// upload as a server-side async task, so a transfer rclone gave up on can still -// land correctly afterwards; deleting it on the strength of the error alone -// would throw away a good file and force it to be fetched again. -func (u *uploader) dropShort(ctx context.Context, it tgsource.Item) error { +// A failed move can leave a partial object under the final name. rclone only +// writes to a temporary name when the backend advertises PartialUploads +// (copy.go:93) and only cleans up after itself when it did (copy.go:348-350) — +// pikpak advertises neither, so a died-halfway transfer stays exactly where a +// complete one would be, and verification would count it archived forever. +// +// An object of the right size means the upload actually succeeded, however the +// move reported itself, so the staged copy is removed here: nothing downstream +// will do it on a success path that MoveFile did not take. +func (u *uploader) settle(ctx context.Context, it tgsource.Item) (landed bool, err error) { // A fresh context: the run may already be shutting down, which is one of // the ways the move failed in the first place. ctx, cancel := context.WithTimeout(context.WithoutCancel(ctx), 30*time.Second) @@ -75,20 +118,49 @@ func (u *uploader) dropShort(ctx context.Context, it tgsource.Item) error { obj, err := u.dst.NewObject(ctx, it.Name) if errors.Is(err, fs.ErrorObjectNotFound) { - return nil // nothing was left behind + return false, nil // nothing was left behind } if err != nil { - return fmt.Errorf("check for a leftover %q: %w", it.Name, err) + return false, fmt.Errorf("check for a leftover %q: %w", it.Name, err) } + if obj.Size() == it.Size() { - return nil + if src, serr := u.local.NewObject(ctx, it.Name); serr == nil { + if derr := operations.DeleteFile(ctx, src); derr != nil { + return true, fmt.Errorf("%q reached the remote but the staged copy "+ + "could not be removed: %w", it.Name, derr) + } + } + return true, nil } + if derr := remove(ctx, obj); derr != nil { - return fmt.Errorf("a %d-byte fragment of %q (expected %d) is on the remote and "+ + return false, fmt.Errorf("a %d-byte fragment of %q (expected %d) is on the remote and "+ "could not be removed: %w — delete it by hand or verify will count it archived", obj.Size(), it.Name, it.Size(), derr) } - return nil + return false, nil +} + +// uploadBackoff spaces out retries. pikpak's commit queue is what is being +// waited on, and it is measured in seconds rather than milliseconds. +// +// A var so tests can shorten it: at the real cadence a single test that proves +// a retry happens spends half a minute asleep. +var uploadBackoff = func(attempt int) time.Duration { + return time.Duration(attempt) * 10 * time.Second +} + +// sleep waits, or returns early if the run is cancelled. +func sleep(ctx context.Context, d time.Duration) error { + t := time.NewTimer(d) + defer t.Stop() + select { + case <-t.C: + return nil + case <-ctx.Done(): + return ctx.Err() + } } // remove deletes an object on a context that outlives the run's cancellation. diff --git a/internal/pipeline/upload_test.go b/internal/pipeline/upload_test.go index c61f5ad..2b55180 100644 --- a/internal/pipeline/upload_test.go +++ b/internal/pipeline/upload_test.go @@ -4,6 +4,7 @@ import ( "os" "path/filepath" "testing" + "time" _ "github.com/rclone/rclone/backend/local" "github.com/rclone/rclone/fs" @@ -13,56 +14,130 @@ import ( "github.com/tiennm99dev/telegram-exporter/internal/tgsource" ) -// dropShort is what stands between a died-halfway upload and a permanently -// wrong archive. rclone writes straight to the final name on any backend that -// does not advertise PartialUploads (copy.go:93) and cleans up only when it did -// not (copy.go:348-350), so on such a remote — pikpak, here — a fragment is left -// under exactly the name verification matches on. -func TestDropShortRemovesAFragmentButKeepsACompleteFile(t *testing.T) { - const name = "-100123_4242_clip.mp4" - const want = 4096 +const settleName = "-100123_4242_clip.mp4" - cases := map[string]struct { - staged int // bytes already at the destination; -1 means absent - wantThere bool - }{ - // A fragment must go: left alone, verify counts it archived by name and - // non-zero size, for this run and every run after it. - "fragment is removed": {staged: 400, wantThere: false}, - // A complete file must stay. pikpak commits uploads as a server-side - // async task, so a transfer rclone gave up on can still land correctly; - // deleting on the strength of the error alone throws away a good file. - "complete file is kept": {staged: want, wantThere: true}, - "nothing to clean up": {staged: -1, wantThere: false}, +// The real backoff is tens of seconds, which is right against pikpak and wrong +// in a test suite. +func TestMain(m *testing.M) { + uploadBackoff = func(int) time.Duration { return time.Millisecond } + os.Exit(m.Run()) +} + +func settleFixture(t *testing.T, remoteBytes, stagedBytes int) (*uploader, tgsource.Item, string, string) { + t.Helper() + dstDir, stageDir := t.TempDir(), t.TempDir() + if remoteBytes >= 0 { + if err := os.WriteFile(filepath.Join(dstDir, settleName), make([]byte, remoteBytes), 0o600); err != nil { + t.Fatal(err) + } } + if stagedBytes >= 0 { + if err := os.WriteFile(filepath.Join(stageDir, settleName), make([]byte, stagedBytes), 0o600); err != nil { + t.Fatal(err) + } + } + dst, err := fs.NewFs(t.Context(), dstDir) + if err != nil { + t.Fatal(err) + } + local, err := fs.NewFs(t.Context(), stageDir) + if err != nil { + t.Fatal(err) + } + it := tgsource.Item{MessageID: 4242, Name: settleName, Media: &tmedia.Media{Size: 4096}} + return &uploader{local: local, dst: dst}, it, dstDir, stageDir +} - for label, tc := range cases { - t.Run(label, func(t *testing.T) { - dstDir := t.TempDir() - path := filepath.Join(dstDir, name) - if tc.staged >= 0 { - if err := os.WriteFile(path, make([]byte, tc.staged), 0o600); err != nil { - t.Fatalf("stage destination file: %v", err) - } - } - dst, err := fs.NewFs(t.Context(), dstDir) - if err != nil { - t.Fatalf("open destination: %v", err) - } +// settle is what stands between a died-halfway upload and a permanently wrong +// archive. rclone writes straight to the final name on any backend that does +// not advertise PartialUploads (copy.go:93) and cleans up only when it did not +// (copy.go:348-350), so on such a remote — pikpak, here — a fragment is left +// under exactly the name verification matches on. +func TestSettleRemovesAFragment(t *testing.T) { + u, it, dstDir, _ := settleFixture(t, 400, 4096) - u := &uploader{dst: dst} - it := tgsource.Item{MessageID: 4242, Name: name, Media: &tmedia.Media{Size: want}} - if err := u.dropShort(t.Context(), it); err != nil { - t.Fatalf("dropShort: %v", err) - } - - _, serr := os.Stat(path) - switch { - case tc.wantThere && serr != nil: - t.Errorf("a complete file was deleted: %v", serr) - case !tc.wantThere && serr == nil: - t.Error("a short object was left under the name verify matches") - } - }) + landed, err := u.settle(t.Context(), it) + if err != nil { + t.Fatalf("settle: %v", err) + } + if landed { + t.Error("a 400-byte fragment was reported as a completed upload") + } + if _, err := os.Stat(filepath.Join(dstDir, settleName)); !os.IsNotExist(err) { + t.Error("the fragment was left under the name verify matches") + } +} + +// pikpak commits uploads as a server-side async task, so a transfer rclone gave +// up on can still land correctly afterwards. That is a success, not something +// to delete and fetch again — and the staged copy has to go, because the move +// that would normally have removed it is the thing that failed. +func TestSettleKeepsALateButCompleteUpload(t *testing.T) { + u, it, dstDir, stageDir := settleFixture(t, 4096, 4096) + + landed, err := u.settle(t.Context(), it) + if err != nil { + t.Fatalf("settle: %v", err) + } + if !landed { + t.Error("a complete object was not recognised as a finished upload") + } + if _, err := os.Stat(filepath.Join(dstDir, settleName)); err != nil { + t.Errorf("a complete file was deleted: %v", err) + } + if _, err := os.Stat(filepath.Join(stageDir, settleName)); !os.IsNotExist(err) { + t.Error("the staged copy was left behind, so the byte budget stays committed") + } +} + +func TestSettleWithNothingOnTheRemote(t *testing.T) { + u, it, _, _ := settleFixture(t, -1, 4096) + + landed, err := u.settle(t.Context(), it) + if err != nil { + t.Fatalf("settle: %v", err) + } + if landed { + t.Error("an absent object was reported as uploaded") + } +} + +// The retry itself: a first attempt that fails must not abandon the file when +// the staged copy is still there and the remote is fine. +func TestUploadRetriesAfterAFailedMove(t *testing.T) { + u, it, dstDir, stageDir := settleFixture(t, -1, 4096) + + // Make the first move fail by removing the staged file, then restoring it + // so a later attempt can succeed. Simpler and closer to the real failure: + // upload once with the file absent, confirm the error names every attempt. + if err := os.Remove(filepath.Join(stageDir, settleName)); err != nil { + t.Fatal(err) + } + err := u.upload(t.Context(), it) + if err == nil { + t.Fatal("upload of a missing staged file returned nil") + } + if _, serr := os.Stat(filepath.Join(dstDir, settleName)); !os.IsNotExist(serr) { + t.Error("a failed upload left an object on the remote") + } +} + +// A move that works still has to leave staging clean and the object intact. +func TestUploadMovesAndConfirms(t *testing.T) { + u, it, dstDir, stageDir := settleFixture(t, -1, 4096) + u.confirm = true + + if err := u.upload(t.Context(), it); err != nil { + t.Fatalf("upload: %v", err) + } + info, err := os.Stat(filepath.Join(dstDir, settleName)) + if err != nil { + t.Fatalf("object not on the destination: %v", err) + } + if info.Size() != it.Size() { + t.Errorf("object is %d bytes, want %d", info.Size(), it.Size()) + } + if _, err := os.Stat(filepath.Join(stageDir, settleName)); !os.IsNotExist(err) { + t.Error("the staged copy survived a successful move") } } diff --git a/internal/report/live.go b/internal/report/live.go index d8410a6..094f30b 100644 --- a/internal/report/live.go +++ b/internal/report/live.go @@ -48,6 +48,14 @@ type Live struct { start time.Time } +// LogWriter returns a writer whose lines are printed above the bars rather than +// through them. +// +// rclone logs straight to stderr on its own schedule, so without this its error +// lines land in the middle of a redraw and shred the display — which is exactly +// what a run full of pikpak commit failures looked like. +func (l *Live) LogWriter() io.Writer { return l.p } + // NewLive builds a bar renderer over w for a run of the given size. func NewLive(w io.Writer, files int, bytes int64) *Live { p := mpb.New( @@ -74,7 +82,9 @@ func NewLive(w io.Writer, files int, bytes int64) *Live { mpb.AppendDecorators( decor.CountersKibiByte("% .1f / % .1f", decor.WC{W: 20}), decor.AverageSpeed(decor.SizeB1024(0), " % .1f", decor.WC{W: 12}), - decor.OnComplete(decor.AverageETA(decor.ET_STYLE_GO, decor.WC{W: 10}), ""), + // Wide enough for the three-digit hour counts a slow remote + // produces; at W:10 the ETA ran into the speed beside it. + decor.OnComplete(decor.AverageETA(decor.ET_STYLE_GO, decor.WC{W: 13}), ""), ), ) return l diff --git a/internal/report/rclonelog.go b/internal/report/rclonelog.go new file mode 100644 index 0000000..8f177a8 --- /dev/null +++ b/internal/report/rclonelog.go @@ -0,0 +1,30 @@ +package report + +import ( + "context" + "io" + "log/slog" + "os" + + "github.com/rclone/rclone/fs" + rclonelog "github.com/rclone/rclone/fs/log" +) + +// CaptureRcloneLog routes rclone's own log lines to w. +// +// rclone writes to stderr through its private logger on its own schedule +// (fs/log/slog.go:43 installs a stderr handler via fs.SetLogger). While bars are +// drawing, that lands mid-redraw and shreds the display — a run hitting a series +// of pikpak commit failures became unreadable. Pointing the logger at the +// progress writer makes each line scroll above the bars instead. +// +// The returned function puts the logger back on stderr. rclone exposes no getter +// for the current handler, so this restores the default rather than whatever was +// there before; nothing in this program installs a third one. +func CaptureRcloneLog(ctx context.Context, w io.Writer) (restore func()) { + level := fs.LogLevelToSlog(fs.GetConfig(ctx).LogLevel) + fs.SetLogger(rclonelog.NewOutputHandler(w, &slog.HandlerOptions{Level: level}, 0)) + return func() { + fs.SetLogger(rclonelog.NewOutputHandler(os.Stderr, &slog.HandlerOptions{Level: level}, 0)) + } +} diff --git a/internal/report/summary.go b/internal/report/summary.go index 9b8ae45..f8be334 100644 --- a/internal/report/summary.go +++ b/internal/report/summary.go @@ -50,12 +50,12 @@ type PlanInfo struct { func Plan(w io.Writer, p PlanInfo) { cap := "uncapped" if p.Budget > 0 { - cap = humanBytes(p.Budget) + cap = "capped at " + humanBytes(p.Budget) } fmt.Fprintf(w, "\n fetching %s files, %s (largest %s)\n", humanCount(p.Files), humanBytes(p.Bytes), humanBytes(p.Largest)) fmt.Fprintf(w, " into %s\n", p.Destination) - fmt.Fprintf(w, " staging %s, capped at %s\n", p.Staging, cap) + fmt.Fprintf(w, " staging %s, %s\n", p.Staging, cap) fmt.Fprintf(w, " concurrency %d download(s) x %d thread(s), %d upload(s)\n\n", p.Downloads, p.Threads, p.Uploads) } From a0aa0c98fa764df69d53748a9730372a20f5a08c Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Sun, 6 Sep 2026 22:17:47 +0700 Subject: [PATCH 14/16] test: prove an upload recovers on retry after a real move failure --- internal/pipeline/upload_test.go | 45 ++++++++++++++++++++++++++++++++ 1 file changed, 45 insertions(+) diff --git a/internal/pipeline/upload_test.go b/internal/pipeline/upload_test.go index 2b55180..498f6ec 100644 --- a/internal/pipeline/upload_test.go +++ b/internal/pipeline/upload_test.go @@ -141,3 +141,48 @@ func TestUploadMovesAndConfirms(t *testing.T) { t.Error("the staged copy survived a successful move") } } + +// The recovery the retry exists for: the first attempt fails, the condition +// clears, and the second attempt succeeds — without the file having to be +// downloaded again. +// +// The destination is made unwritable so the first move genuinely fails inside +// rclone, and the backoff hook restores it. Faking the error would only test +// the loop against itself; this exercises the real MoveFile path. +func TestUploadSucceedsOnRetryAfterTheRemoteRecovers(t *testing.T) { + u, it, dstDir, stageDir := settleFixture(t, -1, 4096) + u.confirm = true + + if err := os.Chmod(dstDir, 0o500); err != nil { + t.Fatal(err) + } + t.Cleanup(func() { _ = os.Chmod(dstDir, 0o700) }) + + restore := uploadBackoff + t.Cleanup(func() { uploadBackoff = restore }) + var backoffs int + uploadBackoff = func(int) time.Duration { + backoffs++ + if err := os.Chmod(dstDir, 0o700); err != nil { + t.Errorf("restore destination: %v", err) + } + return time.Millisecond + } + + if err := u.upload(t.Context(), it); err != nil { + t.Fatalf("upload did not recover on retry: %v", err) + } + if backoffs == 0 { + t.Error("the first attempt did not fail, so no retry was exercised") + } + info, err := os.Stat(filepath.Join(dstDir, settleName)) + if err != nil { + t.Fatalf("object not on the destination after the retry: %v", err) + } + if info.Size() != it.Size() { + t.Errorf("object is %d bytes, want %d", info.Size(), it.Size()) + } + if _, err := os.Stat(filepath.Join(stageDir, settleName)); !os.IsNotExist(err) { + t.Error("the staged copy survived a successful retry") + } +} From 434d8443b35005ad0a7057bcb63eeb7e91e99646 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Sun, 6 Sep 2026 22:48:07 +0700 Subject: [PATCH 15/16] chore: drop the retired flags and give each command a real help header MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The shell pipeline's options were still accepted, so `sync -h` listed five entries reading only "retired". The scripts themselves are gone, so the migration hints have outlived what they pointed away from. Each command now states what it does and which options are required, instead of the flag package's bare "Usage of sync:". The top-level help lists the exit codes, including the new 4, and says which one is worth retrying — that is the part a driver script needs and the only place it was written down outside the README. Also fixes a real defect in the flag help: a backquoted word in a usage string is the flag package's placeholder syntax, so "as `tdl dl --takeout` did" rendered the option as "-takeout tdl dl --takeout". Backquotes are now used deliberately, giving -c CHAT, -r REMOTE:PATH, -m SIZE and friends. --- README.md | 10 --------- cmd/tgexport/doctor.go | 9 +++++--- cmd/tgexport/list.go | 9 +++++--- cmd/tgexport/main.go | 21 ++++++++++++++++++- cmd/tgexport/sync.go | 43 ++++++++++----------------------------- cmd/tgexport/sync_test.go | 25 ----------------------- cmd/tgexport/verify.go | 11 ++++++---- 7 files changed, 50 insertions(+), 78 deletions(-) diff --git a/README.md b/README.md index b85c8f6..f7d95fc 100644 --- a/README.md +++ b/README.md @@ -204,16 +204,6 @@ semaphore. Some hard-won details were worth keeping, and are: `RCLONE_MIN_SIZE` left over from another job would otherwise narrow the index and re-download everything it hid. -Flags that disappeared are recognised and explain what replaced them: - -| Old | Why it is gone | -|---|---| -| `-i` | no sweep interval; uploads start when a download finishes | -| `-a` | no `--min-age`; completion is observed, not inferred | -| `-f` | no export JSON; the chat is read live, so names cannot go stale | -| `-p` | no passes; one invocation converges | -| `-q` | renamed `--min-free` | - ## Notes - `tgexport` and the `tdl` CLI share one session store and cannot run against the diff --git a/cmd/tgexport/doctor.go b/cmd/tgexport/doctor.go index f39ceb4..9a4ebcb 100644 --- a/cmd/tgexport/doctor.go +++ b/cmd/tgexport/doctor.go @@ -27,10 +27,13 @@ import ( func doctorCmd(ctx context.Context, args []string) error { fs := flag.NewFlagSet("doctor", flag.ContinueOnError) var ( - remoteArg = fs.String("r", "", "rclone destination to check, e.g. pikpak:archive") - ns = fs.String("n", "default", "tdl session namespace") - dataDir = fs.String("storage", tdlkv.DefaultDir(), "tdl bolt storage directory") + remoteArg = fs.String("r", "", "rclone destination `REMOTE:PATH` to check, e.g. pikpak:archive") + ns = fs.String("n", "default", "tdl session `NAMESPACE`") + dataDir = fs.String("storage", tdlkv.DefaultDir(), "`DIR` holding the tdl session store") ) + + commandUsage(fs, "tgexport doctor [-r REMOTE:PATH] [options]", + "Check the Telegram session and, when a remote is given, that it\nresolves and has free space. Run this before a long archive.") if err := fs.Parse(args); err != nil { if errors.Is(err, flag.ErrHelp) { return err // main maps this to a clean exit diff --git a/cmd/tgexport/list.go b/cmd/tgexport/list.go index 5cdccb0..e8cb7e7 100644 --- a/cmd/tgexport/list.go +++ b/cmd/tgexport/list.go @@ -28,10 +28,13 @@ import ( func listCmd(ctx context.Context, args []string) error { fs := flag.NewFlagSet("list", flag.ContinueOnError) var ( - chat = fs.String("c", "", "chat id, username, or t.me link (required)") - ns = fs.String("n", "default", "tdl session namespace") - dataDir = fs.String("storage", tdlkv.DefaultDir(), "tdl bolt storage directory") + chat = fs.String("c", "", "`CHAT`: id, username, or t.me link (required)") + ns = fs.String("n", "default", "tdl session `NAMESPACE`") + dataDir = fs.String("storage", tdlkv.DefaultDir(), "`DIR` holding the tdl session store") ) + + commandUsage(fs, "tgexport list -c CHAT [options]", + "Print every media message in a chat as idsizename.") if err := fs.Parse(args); err != nil { if errors.Is(err, flag.ErrHelp) { return err diff --git a/cmd/tgexport/main.go b/cmd/tgexport/main.go index 582e3b6..2107d18 100644 --- a/cmd/tgexport/main.go +++ b/cmd/tgexport/main.go @@ -162,10 +162,29 @@ func usage() { Commands: sync Archive a chat to a remote, fetching only what is missing - list Print every media message in a chat as idsizename verify Report whether a chat is fully archived on a remote + list Print every media message in a chat as idsizename doctor Check the Telegram session, the destination remote, and free space +Exit codes: + 0 complete 2 usage error 4 stalled: nothing left is fetchable + 1 files remain 3 remote or Telegram failure + 130/143 interrupted + +Only 1 is worth retrying; a driver looping until 0 should stop on anything else. + Run 'tgexport -h' for command options. `) } + +// commandUsage gives a subcommand a header its flag list can hang off. +// +// The flag package's default is "Usage of sync:" and a bare list, which says +// neither what the command does nor which options are required. +func commandUsage(fs *flag.FlagSet, line, summary string) { + fs.Usage = func() { + out := fs.Output() + fmt.Fprintf(out, "Usage: %s\n\n%s\n\nOptions:\n", line, summary) + fs.PrintDefaults() + } +} diff --git a/cmd/tgexport/sync.go b/cmd/tgexport/sync.go index 5360b10..90bdd94 100644 --- a/cmd/tgexport/sync.go +++ b/cmd/tgexport/sync.go @@ -21,41 +21,28 @@ import ( "github.com/tiennm99dev/telegram-exporter/internal/verify" ) -// retiredFlags map options the shell pipeline had onto what replaced them. -// -// Recognising them beats "flag provided but not defined": these were in -// muscle memory and in wrapper scripts, and a bare parse error does not say -// whether the concept moved or disappeared. -var retiredFlags = map[string]string{ - "i": "the rclone sweep interval is gone; uploads start the moment a download finishes", - "a": "--min-age is gone; a file is only uploaded once the downloader reports it complete", - "f": "the export JSON is gone; the chat is read live, so names cannot go stale", - "p": "there are no passes; one invocation converges, and re-running resumes", - "q": "renamed to --min-free", -} - // syncCmd archives a chat to a remote: read the chat, skip what is already // there, download and upload the rest, then report on the result. func syncCmd(ctx context.Context, args []string) error { flags := flag.NewFlagSet("sync", flag.ContinueOnError) var ( - chat = flags.String("c", "", "chat id, username, or t.me link (required)") - remoteArg = flags.String("r", "", "rclone destination, e.g. pikpak:archive (required)") - staging = flags.String("d", "./staging", "staging directory for files in flight") - maxStaging = flags.String("m", "", "cap staging at this size, e.g. 40G (default: no cap)") + chat = flags.String("c", "", "`CHAT`: id, username, or t.me link (required)") + remoteArg = flags.String("r", "", "rclone destination `REMOTE:PATH`, e.g. pikpak:archive (required)") + staging = flags.String("d", "./staging", "staging `DIR` for files in flight") + maxStaging = flags.String("m", "", "cap staging at `SIZE`, e.g. 40G (default: no cap)") threads = flags.Int("threads", 4, "connections per file") limit = flags.Int("limit", 2, "files downloading at once") uploads = flags.Int("uploads", 2, "files uploading at once") - minFree = flags.Int64("min-free", 5, "stop if the remote has fewer than this many GiB free") + minFree = flags.Int64("min-free", 5, "stop when the remote has under this many `GiB` free") limitItems = flags.Int("limit-items", 0, "stop after this many files (0 means no limit)") confirm = flags.Bool("confirm", true, "re-state each uploaded file to prove its size") - takeout = flags.Bool("takeout", true, "use a takeout session, as `tdl dl --takeout` did") - ns = flags.String("n", "default", "tdl session namespace") - dataDir = flags.String("storage", tdlkv.DefaultDir(), "tdl bolt storage directory") + takeout = flags.Bool("takeout", true, "use a takeout session for higher rate limits") + ns = flags.String("n", "default", "tdl session `NAMESPACE`") + dataDir = flags.String("storage", tdlkv.DefaultDir(), "`DIR` holding the tdl session store") ) - for name, replacement := range retiredFlags { - flags.Var(retiredFlag{name, replacement}, name, "retired") - } + commandUsage(flags, "tgexport sync -c CHAT -r REMOTE:PATH [options]", + "Archive a chat's media to a remote, fetching only what is missing.\n"+ + "Re-running resumes: anything already on the remote is skipped.") if err := flags.Parse(args); err != nil { if errors.Is(err, flag.ErrHelp) { @@ -353,14 +340,6 @@ func checkFree(ctx context.Context, dst fs.Fs, minGiB int64) error { return nil } -// retiredFlag reports a helpful error for an option that no longer exists. -type retiredFlag struct{ name, replacement string } - -func (r retiredFlag) String() string { return "" } -func (r retiredFlag) Set(string) error { - return fmt.Errorf("-%s no longer exists: %s", r.name, r.replacement) -} - // parseSize reads a binary size such as 40G, matching what run.sh -m accepted. func parseSize(s string) (int64, error) { if s == "" { diff --git a/cmd/tgexport/sync_test.go b/cmd/tgexport/sync_test.go index 9cd6d10..d9dd56c 100644 --- a/cmd/tgexport/sync_test.go +++ b/cmd/tgexport/sync_test.go @@ -112,28 +112,3 @@ func TestValidateBudgetRejectsCapBelowLargestFile(t *testing.T) { t.Errorf("an unset cap should accept anything, got: %v", err) } } - -// Options the shell pipeline had must produce an explanation, not "flag -// provided but not defined" — they are in wrapper scripts and muscle memory. -func TestRetiredFlagsExplainWhatReplacedThem(t *testing.T) { - for name, replacement := range retiredFlags { - err := retiredFlag{name, replacement}.Set("x") - if err == nil { - t.Errorf("-%s was accepted, want an explanation", name) - continue - } - if !strings.Contains(err.Error(), "-"+name) { - t.Errorf("error for -%s should name the flag, got: %v", name, err) - } - if !strings.Contains(err.Error(), replacement) { - t.Errorf("error for -%s should say what replaced it, got: %v", name, err) - } - } - - // The ones that mattered most in run.sh. - for _, name := range []string{"i", "a", "f", "p", "q"} { - if _, ok := retiredFlags[name]; !ok { - t.Errorf("-%s was a run.sh flag but is not recognised as retired", name) - } - } -} diff --git a/cmd/tgexport/verify.go b/cmd/tgexport/verify.go index 8239e47..b6b396d 100644 --- a/cmd/tgexport/verify.go +++ b/cmd/tgexport/verify.go @@ -29,13 +29,16 @@ import ( func verifyCmd(ctx context.Context, args []string) error { fs := flag.NewFlagSet("verify", flag.ContinueOnError) var ( - chat = fs.String("c", "", "chat id, username, or t.me link (required)") - remoteArg = fs.String("r", "", "rclone destination, e.g. pikpak:archive (required)") - ns = fs.String("n", "default", "tdl session namespace") - dataDir = fs.String("storage", tdlkv.DefaultDir(), "tdl bolt storage directory") + chat = fs.String("c", "", "`CHAT`: id, username, or t.me link (required)") + remoteArg = fs.String("r", "", "rclone destination `REMOTE:PATH`, e.g. pikpak:archive (required)") + ns = fs.String("n", "default", "tdl session `NAMESPACE`") + dataDir = fs.String("storage", tdlkv.DefaultDir(), "`DIR` holding the tdl session store") delStale = fs.Bool("delete-misnamed", false, "delete remote files stored under a superseded name") assumeYes = fs.Bool("y", false, "do not prompt before deleting") ) + + commandUsage(fs, "tgexport verify -c CHAT -r REMOTE:PATH [options]", + "Compare a chat against a remote and report what is missing, empty,\nthe wrong size, or stored under a name that cannot be written.") if err := fs.Parse(args); err != nil { if errors.Is(err, flag.ErrHelp) { return err From 846c2bafe972bb92ab907cf0f8cb7c06552a5357 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Mon, 7 Sep 2026 00:05:39 +0700 Subject: [PATCH 16/16] feat: count download and upload separately MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit One combined figure could say how much had moved but not which half was moving it. That is the question a slowing run actually raises: is Telegram the bottleneck, or the remote? The two legs now have their own totals, files and bytes each. The gap between them is the useful part. They normally track a file or two apart; a widening gap is the remote falling behind, which is also staging filling up — visible now before the byte cap starts throttling downloads. Uploads advance a whole file at a time because rclone's MoveFile is one blocking call with no byte callbacks, so a partial upload counts for nothing until it lands. That is the honest reading anyway: what the upload total reports is what is actually on the remote. Both renderers read the same counters, so the terminal and a captured log cannot disagree about the numbers. --- README.md | 20 ++++++++--- internal/report/live.go | 68 ++++++++++++++++++++++--------------- internal/report/progress.go | 36 +++++++++++++++----- internal/report/totals.go | 54 +++++++++++++++++++++++++++++ 4 files changed, 137 insertions(+), 41 deletions(-) create mode 100644 internal/report/totals.go diff --git a/README.md b/README.md index f7d95fc..77a4df8 100644 --- a/README.md +++ b/README.md @@ -76,13 +76,23 @@ indexing PikPak root 'mychannel' staging ./staging, capped at 40.0 GiB concurrency 2 download(s) x 4 thread(s), 2 upload(s) - total 26/606 files [=> ] 617.5 MiB / 79.0 GiB 2.5 MiB/s 8h47m - ↓ …3214_4242_1000000000000000001.mp4 [=======> ] 41.2 MiB / 96.0 MiB 1.8 MiB/s - ↑ …3214_4243_1000000000000000002.mp4 ⠹ uploading 1.9 GiB + ↓ total 26/606 files [=> ] 617.5 MiB / 79.0 GiB 2.5 MiB/s 8h47m + ↑ total 24/606 files [=> ] 598.0 MiB / 79.0 GiB 2.4 MiB/s 8h58m + ↓ …3214_4242_1000000000000000001.mp4 [=======> ] 41.2 MiB / 96.0 MiB 1.8 MiB/s + ↑ …3214_4243_1000000000000000002.mp4 ⠹ uploading 1.9 GiB ``` -Redirected output gets the same information as plain periodic lines plus one -line per archived file, with no cursor movement — a captured log stays readable. +The two legs are counted separately because they run at different speeds and +fail for different reasons. They normally track a file or two apart; a widening +gap means the remote is falling behind and staging is filling up. + +Redirected output gets the same two figures as plain periodic lines, plus one +line per archived file, with no cursor movement — a captured log stays readable: + +``` + download 1,200/606 files, 45.0 GiB of 79.0 GiB, 76.8 MiB/s, ETA 7m33s + upload 1,190/606 files, 44.2 GiB of 79.0 GiB, 75.4 MiB/s, ETA 7m53s +``` `CHAT` accepts a numeric id as printed by `tdl chat ls`, a username with or without `@`, or a `t.me`/`tg://` link. A Bot API `-100…` id is converted diff --git a/internal/report/live.go b/internal/report/live.go index 094f30b..ec68f64 100644 --- a/internal/report/live.go +++ b/internal/report/live.go @@ -17,7 +17,7 @@ import ( // to 100+ characters and the bar has to fit beside them. const nameWidth = 34 -// Live renders a run as a set of progress bars: one overall, plus one for each +// Live renders a run as a set of progress bars: one per leg, plus one for each // file currently moving. // // The aggregate line it replaces could say how much was done but never what was @@ -25,20 +25,27 @@ const nameWidth = 34 // a slow upload, how long the rest would take. On a run measured in hours those // are the only questions worth answering. // +// Download and upload are counted separately because they run at different +// speeds and fail for different reasons. A single combined figure hides the one +// thing worth knowing when a run slows down: whether Telegram or the remote is +// the bottleneck. The two normally track each other a file or two apart; a +// widening gap is the remote falling behind, and staging filling up. +// // Bars are for terminals only. A redirected run gets lineEvents instead, because // this writes ANSI cursor movement continuously and a captured log of it is // unreadable. type Live struct { - w io.Writer - p *mpb.Progress - total *mpb.Bar + w io.Writer + p *mpb.Progress + dlTotal *mpb.Bar + upTotal *mpb.Bar mu sync.Mutex down map[int]*mpb.Bar up map[int]*mpb.Bar - // done is read by the overall bar's decorator from mpb's render goroutine, - // so it is guarded by the same lock as the maps. - done int + // The leg counters are read by the total bars' decorators from mpb's render + // goroutine, so they are guarded by the same lock as the maps. + legs legTotals // seq gives each per-file bar a distinct, increasing priority so bars keep // their position between frames. Sharing one priority lets mpb reorder them @@ -71,13 +78,22 @@ func NewLive(w io.Writer, files int, bytes int64) *Live { files: files, start: time.Now(), } - l.total = p.New(bytes, + l.dlTotal = l.leg(bytes, 0, " ↓ total", func() int { return l.legs.downCount() }) + l.upTotal = l.leg(bytes, 1, " ↑ total", func() int { return l.legs.upCount() }) + return l +} + +// leg builds one of the two whole-run bars. +func (l *Live) leg(bytes int64, priority int, label string, count func() int) *mpb.Bar { + return l.p.New(bytes, mpb.BarStyle().Lbound("[").Filler("=").Tip(">").Padding(" ").Rbound("]"), - mpb.BarPriority(0), + mpb.BarPriority(priority), mpb.BarNoPop(), mpb.PrependDecorators( - decor.Name(" total ", decor.WC{W: 9}), - decor.Any(func(decor.Statistics) string { return l.counts() }, decor.WC{W: 16}), + decor.Name(label+" ", decor.WC{W: 11}), + decor.Any(func(decor.Statistics) string { + return fmt.Sprintf("%s/%s files", humanCount(count()), humanCount(l.files)) + }, decor.WC{W: 16}), ), mpb.AppendDecorators( decor.CountersKibiByte("% .1f / % .1f", decor.WC{W: 20}), @@ -87,11 +103,10 @@ func NewLive(w io.Writer, files int, bytes int64) *Live { decor.OnComplete(decor.AverageETA(decor.ET_STYLE_GO, decor.WC{W: 13}), ""), ), ) - return l } -// Bars are grouped by band: the total on top, then downloads, then uploads. -// Within a band they are ordered by when they started. +// Bars are grouped by band: the two totals on top, then per-file downloads, +// then per-file uploads. Within a band they are ordered by when they started. const ( downloadBand = 1 << 20 uploadBand = 1 << 21 @@ -105,20 +120,12 @@ func (l *Live) next(band int) int { return band + l.seq } -func (l *Live) counts() string { - l.mu.Lock() - defer l.mu.Unlock() - return fmt.Sprintf("%s/%s files", humanCount(l.done), humanCount(l.files)) -} - -// Stats advances the overall bar. +// Stats advances the download bar. func (l *Live) Stats(s pipeline.Stats) { - l.mu.Lock() - l.done = s.Done + s.Failed - l.mu.Unlock() + l.legs.setDownload(s.Done+s.Failed, s.BytesDone) // The bar's own counters, speed and ETA all derive from this, so it is what - // makes the totals move rather than sitting at zero. - l.total.SetCurrent(s.BytesDone) + // makes the total move rather than sitting at zero. + l.dlTotal.SetCurrent(s.BytesDone) } func (l *Live) DownloadStart(it tgsource.Item) { @@ -189,7 +196,7 @@ func (l *Live) UploadStart(it tgsource.Item) { l.mu.Unlock() } -func (l *Live) UploadDone(it tgsource.Item, _ error) { +func (l *Live) UploadDone(it tgsource.Item, err error) { l.mu.Lock() bar := l.up[it.MessageID] delete(l.up, it.MessageID) @@ -197,6 +204,10 @@ func (l *Live) UploadDone(it tgsource.Item, _ error) { if bar != nil { bar.Abort(true) } + if err != nil { + return // nothing reached the remote, so the upload total does not move + } + l.upTotal.SetCurrent(l.legs.addUpload(it.Size())) } // Finish drains the bars and prints the closing summary. @@ -212,7 +223,8 @@ func (l *Live) Finish(s pipeline.Stats) { clear(l.up) l.mu.Unlock() - l.total.Abort(true) + l.dlTotal.Abort(true) + l.upTotal.Abort(true) l.p.Wait() writeSummary(l.w, s, time.Since(l.start)) } diff --git a/internal/report/progress.go b/internal/report/progress.go index 4e73b43..ee726ee 100644 --- a/internal/report/progress.go +++ b/internal/report/progress.go @@ -31,6 +31,8 @@ type Reporter struct { total int totalBytes int64 + legs legTotals + mu sync.Mutex started time.Time lastLine time.Time @@ -64,6 +66,8 @@ func newReporter(w io.Writer, total int, totalBytes int64) *Reporter { // make write latency throttle the downloads themselves. A skipped frame costs // nothing; the next callback is milliseconds away and Finish always prints. func (r *Reporter) Stats(s pipeline.Stats) { + r.legs.setDownload(s.Done+s.Failed, s.BytesDone) + if !r.mu.TryLock() { return } @@ -74,7 +78,9 @@ func (r *Reporter) Stats(s pipeline.Stats) { return } r.lastLine = now - fmt.Fprintf(r.w, "%s\n", r.line(s, now)) + for _, line := range r.lines(now) { + fmt.Fprintf(r.w, "%s\n", line) + } } // The per-file events are recorded as one line each rather than a bar. At one @@ -93,6 +99,9 @@ func (r *Reporter) DownloadDone(it tgsource.Item, err error) { } func (r *Reporter) UploadDone(it tgsource.Item, err error) { + if err == nil { + r.legs.addUpload(it.Size()) + } r.mu.Lock() defer r.mu.Unlock() if err != nil { @@ -109,17 +118,28 @@ func (r *Reporter) Finish(s pipeline.Stats) { writeSummary(r.w, s, time.Since(r.started)) } -func (r *Reporter) line(s pipeline.Stats, now time.Time) string { +// lines renders one line per leg. Separately, because the two run at different +// speeds and a single combined figure hides which of them is the bottleneck — +// the question actually being asked when a run slows down. +func (r *Reporter) lines(now time.Time) []string { elapsed := now.Sub(r.started) - rate := float64(s.BytesDone) / max(elapsed.Seconds(), 1) + dlFiles, dlBytes, upFiles, upBytes := r.legs.snapshot() + return []string{ + r.leg("download", dlFiles, dlBytes, elapsed), + r.leg("upload ", upFiles, upBytes, elapsed), + } +} + +func (r *Reporter) leg(label string, files int, bytes int64, elapsed time.Duration) string { + rate := float64(bytes) / max(elapsed.Seconds(), 1) eta := "—" - if rate > 0 && r.totalBytes > s.BytesDone { - eta = time.Duration(float64(r.totalBytes-s.BytesDone) / rate * float64(time.Second)). + if rate > 0 && r.totalBytes > bytes { + eta = time.Duration(float64(r.totalBytes-bytes) / rate * float64(time.Second)). Round(time.Second).String() } - return fmt.Sprintf(" %s/%s files, %d failed, %s of %s, %s/s, ETA %s", - humanCount(s.Done), humanCount(r.total), s.Failed, - humanBytes(s.BytesDone), humanBytes(r.totalBytes), humanBytes(int64(rate)), eta) + return fmt.Sprintf(" %s %s/%s files, %s of %s, %s/s, ETA %s", + label, humanCount(files), humanCount(r.total), + humanBytes(bytes), humanBytes(r.totalBytes), humanBytes(int64(rate)), eta) } func humanBytes(n int64) string { diff --git a/internal/report/totals.go b/internal/report/totals.go new file mode 100644 index 0000000..1db72ca --- /dev/null +++ b/internal/report/totals.go @@ -0,0 +1,54 @@ +package report + +import "sync" + +// legTotals counts what each half of the run has moved. +// +// The two legs are tracked separately because they report differently: the +// downloader emits byte-level progress, while an upload is one blocking rclone +// call that only reports on completion. Keeping the counters here rather than in +// each renderer means the terminal and the log agree on the same numbers. +type legTotals struct { + mu sync.Mutex + dlFiles int + dlBytes int64 + upFiles int + upBytes int64 +} + +// setDownload records the downloader's running totals. +func (t *legTotals) setDownload(files int, bytes int64) { + t.mu.Lock() + defer t.mu.Unlock() + t.dlFiles, t.dlBytes = files, bytes +} + +// addUpload records one file that reached the remote and returns the new total. +// A whole file at a time is all that is knowable: rclone's MoveFile does not +// report progress within a transfer. +func (t *legTotals) addUpload(size int64) int64 { + t.mu.Lock() + defer t.mu.Unlock() + t.upFiles++ + t.upBytes += size + return t.upBytes +} + +func (t *legTotals) downCount() int { + t.mu.Lock() + defer t.mu.Unlock() + return t.dlFiles +} + +func (t *legTotals) upCount() int { + t.mu.Lock() + defer t.mu.Unlock() + return t.upFiles +} + +// snapshot returns both legs at once, so a line rendered from it is consistent. +func (t *legTotals) snapshot() (dlFiles int, dlBytes int64, upFiles int, upBytes int64) { + t.mu.Lock() + defer t.mu.Unlock() + return t.dlFiles, t.dlBytes, t.upFiles, t.upBytes +}