From 5e7630bf5cbf56a846d0f5da929bf838513633f3 Mon Sep 17 00:00:00 2001 From: dat972 Date: Mon, 24 Aug 2026 21:43:43 -0500 Subject: [PATCH] updated the reorganize script to handle nested folders inside of template files like 'YYYY' Income Documents > 2016 > Joe Shmo 2016 --- README.md | 2 ++ fccs_reorganize.py | 28 ++++++++++++++++++++++------ fccs_report_reorganize.py | 9 ++++++++- 3 files changed, 32 insertions(+), 7 deletions(-) diff --git a/README.md b/README.md index 68d9c04..5f525f7 100644 --- a/README.md +++ b/README.md @@ -97,6 +97,8 @@ output/ Exported filenames follow the format `{drawer_id}_{client_name}_{folder_name}_{creation_date}_{document_name}.ext`. The parser uses folder templates from `fccs_folders.txt` (with `YYYY` expanded via regex) and the creation date (`MM-DD-YYYY`) as anchors to reliably split the underscore-delimited fields. Folder names are decomposed into nested paths that match the FCCS UI structure (e.g. `2025 Tax Documents` becomes `Tax Documents/2025/`). UltraTax CS folders are matched by a built-in pattern. +FCCS drawers can (rarely) contain **nested subfolders** under a template folder; the export encodes these in square brackets appended to the folder field, e.g. `2016 Income Documents[4201 N Beach Street, LLC]`. The reorganizer recreates them as deeper nesting — that example becomes `Income Documents/2016/4201 N Beach Street, LLC/` — as long as the parent (`YYYY Income Documents`) is a listed template; no bracket entries are needed in `fccs_folders.txt`. + Files that can't be fully parsed still keep their client: the drawer ID and client name are the first two underscore-delimited tokens and stay recoverable even when the folder/date parse fails, so those files are filed under `{client_name}/_unparsed/` (retaining their original filename). Only files whose client can't be recovered at all fall back to the top-level `_unparsed/`. Folder names are sanitized for Windows before use — trailing spaces and periods are stripped from each path component, since Windows can't create a directory ending in a space or dot (e.g. a client named `FARR GROUP, P.L.` becomes `FARR GROUP, P.L`). Reorganization is also resilient per file: if one file can't be placed for any reason, the error is logged and counted (reported as `errored` in the summary) and the run continues with the rest rather than aborting. diff --git a/fccs_reorganize.py b/fccs_reorganize.py index 9f68d9d..7ac7889 100644 --- a/fccs_reorganize.py +++ b/fccs_reorganize.py @@ -36,6 +36,10 @@ from fccs_config import ( # Regex for the creation date field (MM-DD-YYYY) bounded by underscores _DATE_RE = re.compile(r"_(\d{2}-\d{2}-\d{4})_") +# Trailing "[Subfolder Name]" on the folder field — FCCS encodes nested +# subfolders (e.g. Income Documents > 2016 > "4201 N Beach Street, LLC") as +# "2016 Income Documents[4201 N Beach Street, LLC]" in the export filename. +_SUBFOLDER_RE = re.compile(r"\[([^\[\]]+)\]$") def parse_filename(filename, folder_patterns): @@ -81,17 +85,29 @@ def parse_filename(filename, folder_patterns): after_date_pos = m.start() - 1 + len("_") + len(date_str) + len("_") doc_name = rest[after_date_pos:] - # Try each folder pattern against the end of before_date + # Peel any trailing [Subfolder] groups off the folder field so the + # remainder can match a template; each becomes one more nested path + # component (in order, so "a[b][c]" nests as a/b/c). + folder_text = before_date + sub_parts = [] + while True: + sm = _SUBFOLDER_RE.search(folder_text) + if not sm: + break + sub_parts.insert(0, sm.group(1)) + folder_text = folder_text[: sm.start()] + + # Try each folder pattern against the end of the folder text for pattern, template_name, folder_type in folder_patterns: - folder_match = pattern.search(before_date) - if folder_match and folder_match.end() == len(before_date): + folder_match = pattern.search(folder_text) + if folder_match and folder_match.end() == len(folder_text): # Folder matched at the end — check for underscore separator before it folder_start = folder_match.start() - if folder_start > 0 and before_date[folder_start - 1] == "_": - client_name = before_date[: folder_start - 1] + if folder_start > 0 and folder_text[folder_start - 1] == "_": + client_name = folder_text[: folder_start - 1] folder_parts = decompose_folder_path( folder_match, folder_type, template_name - ) + ) + tuple(sub_parts) return (drawer_id, client_name, folder_parts, date_str, doc_name + ext) return None diff --git a/fccs_report_reorganize.py b/fccs_report_reorganize.py index bb61400..8d885b9 100644 --- a/fccs_report_reorganize.py +++ b/fccs_report_reorganize.py @@ -26,6 +26,8 @@ from fccs_config import parse_args, load_config _DATE_RE = re.compile(r"_(\d{2}-\d{2}-\d{4})_") # A standalone year inside a folder name, e.g. "2013 Tax Documents". _YEAR_RE = re.compile(r"\b(19|20)\d{2}\b") +# Trailing "[Subfolder]" group(s) — FCCS's encoding for nested subfolders. +_SUBFOLDER_RE = re.compile(r"(?:\[[^\[\]]+\])+$") UNPARSED_DIRNAME = "_unparsed" @@ -55,8 +57,13 @@ def suggest_template(folder_field): """Generalize a concrete folder field into an fccs_folders.txt line. e.g. "2013 Tax Documents" -> "YYYY Tax Documents". Fields without a year - are suggested as-is (static folders). + are suggested as-is (static folders). Trailing [Subfolder] groups (FCCS's + nested-subfolder encoding) are dropped so the suggestion is the parent + template line, which is what fccs_folders.txt actually takes. """ + stripped = _SUBFOLDER_RE.sub("", folder_field).strip() + if stripped: + folder_field = stripped return _YEAR_RE.sub("YYYY", folder_field)