feat: estimate the time remaining from bytes and observed rate
Build and publish container / build (pull_request) Successful in 2m13s

rsync reports each file's size with %l as it completes, which is all an
estimate needs: bytes done over time elapsed is the same arithmetic rsync would
do internally, and requires nothing it does not already print. The scan pass now
sums those sizes as well as counting files, so both a percentage and an estimate
have a real denominator.

The rate is measured over a trailing thirty seconds rather than the whole run,
so it follows a device that slows down instead of averaging the slowdown away --
which for a card reader that thermally throttles, or a USB link that renegotiates
after an hour, is the difference between a useful estimate and a reassuring one.

Below two seconds no rate is reported at all. The first handful of files arrive
microseconds apart, and dividing by that window produces a rate in the gigabytes
per second and an estimate of zero, which is worse than showing nothing.

Directory entries are excluded from the byte total as well as the file count.
rsync reports them with a 4096 inode size, which across six thousand album
directories is several megabytes of transfer that never happens.
This commit is contained in:
Emma Thorpe
2026-08-25 12:40:32 +01:00
parent 37b841f009
commit 131c80f5de
4 changed files with 219 additions and 27 deletions
+111 -10
View File
@@ -1,7 +1,10 @@
#!/usr/bin/env python3
"""Render rsync's per-file output as a single updating status line.
Fed the paths rsync reports with --out-format='%n', one per line. Prints one
Fed the size and path rsync reports with --out-format='%l %n', one per line.
The size is what makes an estimate possible: rsync's own rate is not exposed
per file, but bytes completed over time elapsed is the same arithmetic and
needs nothing rsync does not already print. Prints one
line that rewrites itself, showing how far through the transfer is and which
album is currently going across, rather than either scrolling fifty thousand
filenames past or -- as rsync does while it builds its file list -- saying
@@ -12,11 +15,76 @@ not fill up with carriage returns.
"""
import argparse
import collections
import os
import shutil
import sys
import time
# The estimate is taken over a trailing window rather than the whole run, so it
# follows a device that slows down instead of averaging the slowdown away.
RATE_WINDOW_SECONDS = 30.0
# Below this the window is too narrow to divide by: the first few files arrive
# in microseconds and produce a rate in the gigabytes per second, and an ETA of
# nothing at all. Better to show neither until the figure means something.
RATE_MINIMUM_SPAN_SECONDS = 2.0
def parse(line):
"""Return (bytes, path) for one line of rsync output.
Tolerates a bare path, in case someone runs this against --out-format='%n'.
"""
line = line.rstrip("\n")
size, separator, path = line.partition(" ")
if separator and size.isdigit():
return int(size), path
return 0, line
def human_bytes(count):
size = float(count)
for unit in ("B", "KiB", "MiB", "GiB", "TiB"):
if size < 1024 or unit == "TiB":
return f"{size:.1f} {unit}"
size /= 1024
def human_duration(seconds):
"""Return a duration nobody has to do arithmetic on."""
seconds = int(seconds)
if seconds < 60:
return f"{seconds}s"
if seconds < 3600:
return f"{seconds // 60}m{seconds % 60:02d}s"
return f"{seconds // 3600}h{(seconds % 3600) // 60:02d}m"
class Rate:
"""Bytes per second over a trailing window."""
def __init__(self, window=RATE_WINDOW_SECONDS):
self.window = window
self.samples = collections.deque()
def add(self, when, total_bytes):
self.samples.append((when, total_bytes))
while len(self.samples) > 2 and when - self.samples[0][0] > self.window:
self.samples.popleft()
def per_second(self):
if len(self.samples) < 2:
return 0.0
(first_time, first_bytes), (last_time, last_bytes) = (
self.samples[0],
self.samples[-1],
)
elapsed = last_time - first_time
if elapsed < RATE_MINIMUM_SPAN_SECONDS:
return 0.0
return (last_bytes - first_bytes) / elapsed
def album_of(path):
"""Return "Artist / Album" for a mirror-relative path."""
@@ -35,18 +103,29 @@ def fit(text, width):
return "" + text[-(width - 1) :]
def render(done, total, label, width):
def render(done, total, copied, expected, rate, label, width):
"""Build the status line, giving whatever room is left to the album."""
if total > 0:
share = min(100, done * 100 // total)
head = f"[{done:>6,} / {total:<6,} {share:>3}%] "
head = f"[{share:>3}%] {done:,}/{total:,}"
else:
head = f"[{done:>6,} files] "
head = f"[{done:,} files]"
if expected > 0:
head += f" {human_bytes(copied)}/{human_bytes(expected)}"
if rate > 0:
head += f" {human_bytes(rate)}/s"
remaining = expected - copied
if remaining > 0:
head += f" ETA {human_duration(remaining / rate)}"
head += " "
return head + fit(label, max(0, width - len(head)))
def main(argv=None, stream=None, out=None):
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--total", type=int, default=0, help="files expected")
parser.add_argument("--bytes", type=int, default=0, help="bytes expected")
parser.add_argument("--interval", type=float, default=0.1, help="seconds between redraws")
args = parser.parse_args(argv)
@@ -56,31 +135,53 @@ def main(argv=None, stream=None, out=None):
width = shutil.get_terminal_size((100, 24)).columns - 1
done = 0
copied = 0
rate = Rate()
started = time.monotonic()
rate.add(started, 0)
last_drawn = 0.0
label = ""
for line in stream:
path = line.rstrip("\n")
# rsync reports directories too, with a trailing slash. They are not
# files and counting them would put the percentage past a hundred.
size, path = parse(line)
# rsync reports directories too, with a trailing slash and an inode
# size. Counting them puts the percentage past a hundred and the byte
# total well over what will actually be transferred.
if not path or path.endswith("/"):
continue
done += 1
copied += size
label = album_of(path) or os.path.basename(path)
now = time.monotonic()
rate.add(now, copied)
if interactive:
if now - last_drawn >= args.interval:
out.write("\r\033[2K" + render(done, args.total, label, width))
out.write(
"\r\033[2K"
+ render(done, args.total, copied, args.bytes, rate.per_second(),
label, width)
)
out.flush()
last_drawn = now
elif now - last_drawn >= 30:
out.write(render(done, args.total, label, width) + "\n")
out.write(
render(done, args.total, copied, args.bytes, rate.per_second(), label, width)
+ "\n"
)
out.flush()
last_drawn = now
elapsed = max(1e-9, time.monotonic() - started)
if interactive:
out.write("\r\033[2K")
out.write(render(done, args.total, label, width).rstrip() + "\n")
summary = render(done, args.total, copied, args.bytes, 0, label, width).rstrip()
out.write(f"{summary}\n")
if copied:
out.write(
f"copied {human_bytes(copied)} in {human_duration(elapsed)}"
f" at {human_bytes(copied / elapsed)}/s\n"
)
out.flush()
return 0