diff --git a/helpers/rerun_failed_uploads.py b/helpers/rerun_failed_uploads.py new file mode 100755 index 0000000..ccd61c1 --- /dev/null +++ b/helpers/rerun_failed_uploads.py @@ -0,0 +1,374 @@ +#!/usr/bin/env python3 +""" +rerun_failed_uploads.py - parse a SmartVault bulk-upload log, work out which +files still need uploading, and optionally copy them (mirroring the relative +folder structure) to a new root so the upload tool can be re-run against it. + +Default behaviour is a DRY RUN: it prints a summary and writes a report. +Add --copy to actually copy files. + +What counts as "needs re-upload": + * http-error : a file line in the error section (502/503/504 etc.) + * folder-error : "All files under subdirectory: X [Can't Upload Folder]" + -> files sitting DIRECTLY in X (non-recursive) + * after-crash : the run aborted; the tool walks ~alphabetically, so every + top-level folder sorting >= the crash file's top-level folder + (inclusive) is included entirely. Disable with --no-after-crash. + +Files in the "renamed due to invalid character" section DID upload and are +only reported, never copied. + +Paths inside the log are always Windows paths (G:\\...). They are parsed with +ntpath regardless of the OS this script runs on, and the drive letter is +translated to a local mount via DRIVE_MAP (see below) so the script works on +Linux with the drive mounted under /run/media. +""" +from __future__ import annotations + +import argparse +import ntpath +import os +import re +import shutil +import sys +from dataclasses import dataclass, field + +# --------------------------------------------------------------------------- +# MACHINE-SPECIFIC SETTINGS - EDIT THESE WHEN MOVING BETWEEN LINUX AND WINDOWS +# --------------------------------------------------------------------------- +# The log file always contains Windows paths (G:\Source_Files_Organized\...). +# DRIVE_MAP tells the script where that drive letter is actually mounted on the +# machine running this script. Only the drive letter is translated; the rest of +# the path is kept as-is. +# +# Linux (Fedora, drive auto-mounted under /run/media): +# DRIVE_MAP = {"G:": "/run/media/way/Kirk H Fritschen CPA"} +# +# Windows (drive mapped as G:): +# DRIVE_MAP = {"G:": "G:\\"} +# +# Instead of editing this you can also pass --source-root on the command line, +# e.g. --source-root "G:\Source_Files_Organized" on Windows. +DRIVE_MAP = { + "G:": "/run/media/way/Kirk H Fritschen CPA", +} + +# Where the files that need re-uploading get mirrored to (relative folder +# structure is preserved underneath it). It is kept on the same drive so the +# Windows upload tool can afterwards be pointed at G:\Rerun_Source_Files. +# +# Linux: DEFAULT_DEST_ROOT = "/run/media/way/Kirk H Fritschen CPA/Rerun_Source_Files" +# Windows: DEFAULT_DEST_ROOT = r"G:\Rerun_Source_Files" +# +# Can also be overridden per run with --dest-root. +DEFAULT_DEST_ROOT = "/run/media/way/Kirk H Fritschen CPA/Rerun_Source_Files" +# --------------------------------------------------------------------------- + +HDR_RENAMED = "contain an invalid character" +HDR_ERRORS = "were skipped due to an error" +HDR_LASTFILE = "Last file uploaded was:" + +RE_SUBDIR = re.compile(r"^All files under subdirectory:\s+(?P.+?)\s{2,}\[", re.I) +RE_UPLOADSET = re.compile(r"^Upload Set#:", re.I) +# Path is greedy: file/folder names may themselves contain double spaces, so the +# split must happen at the LAST run of 2+ spaces. The message never contains a +# backslash or a double space. +RE_FILE_ERR = re.compile(r"^(?P(?:[A-Za-z]:\\|\\\\).+)\s{2,}(?P[^\\]+?)\s*$") +RE_RENAMED = re.compile(r"^(?P.+?)\s{2,}\[(?Pfile|dir\w*)\]\s*$", re.I) + + +@dataclass +class LogData: + last_file: str | None = None + renamed: list[str] = field(default_factory=list) + failed_files: list[tuple[str, str]] = field(default_factory=list) # (path, msg) + failed_dirs: list[str] = field(default_factory=list) + upload_set_errors: int = 0 + unparsed: list[str] = field(default_factory=list) + + +def parse_log(log_path: str) -> LogData: + data = LogData() + section = None # None | "renamed" | "errors" + expect_last = False + with open(log_path, encoding="utf-8", errors="replace") as fh: + for raw in fh: + line = raw.rstrip("\r\n") + s = line.strip() + if not s: + continue + if expect_last: + data.last_file = s + expect_last = False + continue + if s.startswith(HDR_LASTFILE): + expect_last = True + continue + if HDR_RENAMED in s: + section = "renamed" + continue + if HDR_ERRORS in s: + section = "errors" + continue + if not line.startswith(" "): # other summary/header lines + continue + + if section == "renamed": + m = RE_RENAMED.match(s) + if m: + data.renamed.append(m.group("path")) + else: + data.unparsed.append(s) + elif section == "errors": + m = RE_SUBDIR.match(s) + if m: + data.failed_dirs.append(m.group("path")) + continue + if RE_UPLOADSET.match(s): + data.upload_set_errors += 1 + continue + m = RE_FILE_ERR.match(s) + if m: + data.failed_files.append((m.group("path"), m.group("msg"))) + else: + data.unparsed.append(s) + return data + + +def detect_source_root(data: LogData) -> str | None: + paths = [p for p, _ in data.failed_files] + data.failed_dirs + data.renamed + if data.last_file: + paths.append(data.last_file) + if not paths: + return None + # Directory-level common prefix (path components, case-insensitive). + # Log paths are Windows paths -> always use ntpath, whatever the host OS. + split = [ntpath.normpath(p).split("\\") for p in paths] + common: list[str] = [] + for parts in zip(*split): + if all(x.casefold() == parts[0].casefold() for x in parts): + common.append(parts[0]) + else: + break + if not common: + return None + root = "\\".join(common) + if len(common) == 1: # "G:" -> "G:\\" + root += "\\" + return root + + +def log_path_to_local(path: str) -> str: + """Translate a Windows path from the log (G:\\a\\b) to a local path using DRIVE_MAP.""" + drive, rest = ntpath.splitdrive(ntpath.normpath(path)) + base = DRIVE_MAP.get(drive.upper()) + if base is None: + if os.name == "nt": + base = drive + "\\" + else: + raise SystemExit(f"ERROR: no DRIVE_MAP entry for drive '{drive}' (path: {path})") + parts = [x for x in rest.split("\\") if x] + return os.path.join(base, *parts) if parts else base + + +def _norm(p: str) -> str: + return os.path.normcase(os.path.normpath(p)) + + +def _remap(path: str, log_root: str, real_root: str) -> str: + """Translate a Windows path from the log (under log_root) to the real local root.""" + rel = ntpath.relpath(ntpath.normpath(path), ntpath.normpath(log_root)) + if rel == ".": + return real_root + return os.path.join(real_root, *rel.split("\\")) + + +def resolve_targets(data: LogData, log_root: str, source_root: str, + include_after_crash: bool, warn) -> dict[str, str]: + targets: dict[str, str] = {} # normcase path -> reason + display: dict[str, str] = {} # normcase path -> original-case path + + def add(path: str, reason: str) -> None: + key = _norm(path) + if key not in targets: + targets[key] = reason + display[key] = os.path.normpath(path) + + # 1. single-file errors + for p, _msg in data.failed_files: + rp = _remap(p, log_root, source_root) + if os.path.isfile(rp): + add(rp, "http-error") + else: + warn(f"missing file (http-error): {rp}") + + # 2. folder errors, non-recursive + for d in dict.fromkeys(data.failed_dirs): # unique, keep order + rd = _remap(d, log_root, source_root) + if not os.path.isdir(rd): + warn(f"missing folder (folder-error): {rd}") + continue + try: + with os.scandir(rd) as it: + for e in it: + if e.is_file(follow_symlinks=False): + add(e.path, "folder-error") + except OSError as ex: + warn(f"cannot scan {rd}: {ex}") + + # 3. everything from the crash point onward + if include_after_crash and data.last_file: + rel = ntpath.relpath(ntpath.normpath(data.last_file), ntpath.normpath(log_root)) + crash_top = rel.split("\\")[0] + if crash_top in (".", ".."): + warn("crash file is not under the source root; skipping after-crash inclusion") + elif not os.path.isdir(source_root): + warn(f"source root not found, cannot enumerate after-crash folders: {source_root}") + else: + ct = crash_top.casefold() + tops = sorted( + (e for e in os.scandir(source_root) if e.is_dir(follow_symlinks=False)), + key=lambda e: e.name.casefold(), + ) + chosen = [e for e in tops if e.name.casefold() >= ct] + for e in chosen: + for dirpath, _dirs, files in os.walk(e.path): + for f in files: + add(os.path.join(dirpath, f), "after-crash") + warn(f"after-crash: crash top-level folder = '{crash_top}', " + f"{len(chosen)} of {len(tops)} top-level folders included") + + return {display[k]: v for k, v in targets.items()} + + +def _long(p: str) -> str: + if os.name == "nt" and len(p) > 240 and not p.startswith("\\\\?\\"): + p = os.path.abspath(p) + return "\\\\?\\UNC\\" + p[2:] if p.startswith("\\\\") else "\\\\?\\" + p + return p + + +def copy_targets(targets: dict[str, str], source_root: str, dest_root: str, + do_copy: bool, skip_existing: bool): + copied = skipped = 0 + errors: list[tuple[str, str]] = [] + total = len(targets) + for i, src in enumerate(targets, 1): + rel = os.path.relpath(src, source_root) + dst = os.path.join(dest_root, rel) + if not do_copy: + continue + try: + if skip_existing and os.path.isfile(dst) and \ + os.path.getsize(dst) == os.path.getsize(src): + skipped += 1 + continue + os.makedirs(_long(os.path.dirname(dst)), exist_ok=True) + shutil.copy2(_long(src), _long(dst)) + copied += 1 + except OSError as ex: + errors.append((src, str(ex))) + if i % 500 == 0 or i == total: + print(f" [{i}/{total}] copied={copied} skipped={skipped} errors={len(errors)}", + flush=True) + return copied, skipped, errors + + +def main(argv=None) -> int: + ap = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("log", help="SmartVault upload log file") + ap.add_argument("--source-root", help="LOCAL path of the root the tool was pointed at " + "(default: auto-detected from the log paths, translated via DRIVE_MAP)") + ap.add_argument("--dest-root", default=DEFAULT_DEST_ROOT, + help="where to mirror the files to (default: %(default)s)") + ap.add_argument("--copy", action="store_true", + help="actually copy files (default is dry run)") + ap.add_argument("--no-after-crash", action="store_true", + help="do NOT include folders from the crash point onward") + ap.add_argument("--skip-existing", action="store_true", + help="skip files already in dest with the same size") + ap.add_argument("--report", help="report path (default: rerun_report.txt next to the log)") + args = ap.parse_args(argv) + + warnings: list[str] = [] + + def warn(msg: str) -> None: + warnings.append(msg) + print("WARN:", msg, file=sys.stderr) + + data = parse_log(args.log) + log_root = detect_source_root(data) + if log_root is None: + print("ERROR: no paths found in log; cannot determine source root", file=sys.stderr) + return 2 + source_root = os.path.normpath(args.source_root) if args.source_root \ + else log_path_to_local(log_root) + dest_root = os.path.normpath(args.dest_root) + report_path = args.report or os.path.join( + os.path.dirname(os.path.abspath(args.log)), "rerun_report.txt") + + print(f"Log : {args.log}") + print(f"Log root : {log_root}") + print(f"Source root : {source_root}") + print(f"Dest root : {dest_root}") + print(f"Crash file : {data.last_file}") + print(f"Parsed : {len(data.failed_files)} file errors, " + f"{len(data.failed_dirs)} folder errors ({len(set(map(_norm, data.failed_dirs)))} unique), " + f"{len(data.renamed)} renamed (uploaded OK), " + f"{data.upload_set_errors} upload-set lines ignored, " + f"{len(data.unparsed)} unparsed lines") + for u in data.unparsed[:10]: + print(" unparsed:", u) + + targets = resolve_targets(data, log_root, source_root, + not args.no_after_crash, warn) + + by_reason: dict[str, int] = {} + total_bytes = 0 + for p, r in targets.items(): + by_reason[r] = by_reason.get(r, 0) + 1 + try: + total_bytes += os.path.getsize(p) + except OSError: + pass + + print(f"\nFiles to re-upload: {len(targets)} ({total_bytes / 1e9:.3f} GB)") + for r, n in sorted(by_reason.items()): + print(f" {r:13s} {n}") + + copied = skipped = 0 + errors: list[tuple[str, str]] = [] + if args.copy: + print(f"\nCopying to {dest_root} ...") + copied, skipped, errors = copy_targets(targets, source_root, dest_root, + True, args.skip_existing) + print(f"Done: copied={copied} skipped={skipped} errors={len(errors)}") + else: + print("\nDRY RUN - nothing copied. Re-run with --copy to copy.") + + with open(report_path, "w", encoding="utf-8") as fh: + fh.write(f"log: {args.log}\nsource_root: {source_root}\ndest_root: {dest_root}\n") + fh.write(f"crash_file: {data.last_file}\nmode: {'COPY' if args.copy else 'DRY RUN'}\n") + fh.write(f"files_to_reupload: {len(targets)}\ntotal_bytes: {total_bytes}\n") + for r, n in sorted(by_reason.items()): + fh.write(f" {r}: {n}\n") + fh.write(f"renamed_uploaded_ok: {len(data.renamed)}\n") + fh.write(f"upload_set_lines_ignored: {data.upload_set_errors}\n") + if args.copy: + fh.write(f"copied: {copied}\nskipped: {skipped}\ncopy_errors: {len(errors)}\n") + if warnings: + fh.write("\n# warnings\n") + fh.writelines(f"{w}\n" for w in warnings) + if errors: + fh.write("\n# copy errors\n") + fh.writelines(f"{p}\t{e}\n" for p, e in errors) + fh.write("\n# targets (reasonpath)\n") + for p, r in sorted(targets.items(), key=lambda kv: kv[0].casefold()): + fh.write(f"{r}\t{p}\n") + print(f"Report written : {report_path}") + return 1 if errors else 0 + + +if __name__ == "__main__": + sys.exit(main())