311 lines
10 KiB
Python
311 lines
10 KiB
Python
"""
|
|
Shared configuration, logging, and utilities for the FCCS extraction tool.
|
|
"""
|
|
|
|
import os
|
|
import re
|
|
import sys
|
|
import argparse
|
|
import configparser
|
|
from datetime import datetime
|
|
|
|
DEFAULT_CONFIG = "config.ini"
|
|
|
|
|
|
def parse_args():
|
|
"""Parse the optional --config argument common to all scripts."""
|
|
parser = argparse.ArgumentParser()
|
|
parser.add_argument(
|
|
"--config",
|
|
default=None,
|
|
help=f"Path to config file (default: {DEFAULT_CONFIG} in script directory)",
|
|
)
|
|
return parser.parse_args()
|
|
|
|
|
|
def load_config(path=None):
|
|
"""Read config.ini and return a ConfigParser object."""
|
|
if path is None:
|
|
script_dir = os.path.dirname(os.path.abspath(sys.argv[0]))
|
|
path = os.path.join(script_dir, DEFAULT_CONFIG)
|
|
if not os.path.exists(path):
|
|
print(f"ERROR: config file not found: {path}")
|
|
sys.exit(1)
|
|
cfg = configparser.ConfigParser()
|
|
cfg.read(path, encoding="utf-8")
|
|
return cfg
|
|
|
|
|
|
def make_logger(log_file):
|
|
"""Return a log() function that writes timestamped lines to console and file."""
|
|
def log(msg):
|
|
line = f"[{datetime.now():%Y-%m-%d %H:%M:%S}] {msg}"
|
|
print(line)
|
|
try:
|
|
with open(log_file, "a", encoding="utf-8") as f:
|
|
f.write(line + "\n")
|
|
except Exception:
|
|
pass
|
|
return log
|
|
|
|
|
|
def load_folder_list(path):
|
|
"""Load known FCCS folder names from file, one per line."""
|
|
if not os.path.exists(path):
|
|
print(f"ERROR: folder list file not found: {path}")
|
|
sys.exit(1)
|
|
folders = []
|
|
with open(path, "r", encoding="utf-8") as f:
|
|
for line in f:
|
|
s = line.strip()
|
|
if s and not s.startswith("#"):
|
|
folders.append(s)
|
|
return folders
|
|
|
|
|
|
def build_folder_patterns(templates):
|
|
"""
|
|
Convert folder template strings into regex patterns for matching.
|
|
|
|
Templates use YYYY as a year placeholder (e.g. 'YYYY Tax Documents').
|
|
Non-recurring folders (no YYYY) are matched literally.
|
|
A built-in pattern for 'UltraTax CS MM-DD-YYYY' is always included.
|
|
|
|
Returns a list of (compiled_regex, template_name, folder_type) tuples,
|
|
sorted longest-first to prevent partial matches.
|
|
|
|
folder_type is one of:
|
|
"yyyy" — recurring folder with year prefix (captured in group 1)
|
|
"ultratax" — UltraTax CS folder with date suffix (captured in group 1)
|
|
"static" — non-recurring folder, no decomposition needed
|
|
"""
|
|
patterns = []
|
|
|
|
# Built-in: UltraTax CS folders (auto-generated by UltraTax integration)
|
|
patterns.append((
|
|
re.compile(r"UltraTax CS (\d{2}-\d{2}-\d{4})$"),
|
|
"UltraTax CS",
|
|
"ultratax",
|
|
))
|
|
|
|
for tmpl in templates:
|
|
if "YYYY" in tmpl:
|
|
# Replace YYYY with captured 4-digit year pattern, escape the rest
|
|
parts = tmpl.split("YYYY")
|
|
regex_str = re.escape(parts[0]) + r"(\d{4})" + re.escape(parts[1])
|
|
patterns.append((re.compile(regex_str + "$"), tmpl, "yyyy"))
|
|
else:
|
|
# Non-recurring folder — exact match
|
|
regex_str = re.escape(tmpl)
|
|
patterns.append((re.compile(regex_str + "$"), tmpl, "static"))
|
|
|
|
# Sort by regex pattern length (longest first) to avoid partial matches
|
|
patterns.sort(key=lambda p: len(p[0].pattern), reverse=True)
|
|
return patterns
|
|
|
|
|
|
def decompose_folder_path(match, folder_type, template_name):
|
|
"""
|
|
Decompose a matched folder name into nested path components that
|
|
recreate the FCCS UI folder structure.
|
|
|
|
Returns a tuple of path parts to be joined with os.path.join().
|
|
|
|
Examples:
|
|
yyyy: "YYYY Tax Documents" matched "2025 Tax Documents"
|
|
→ ("Tax Documents", "2025")
|
|
ultratax: "UltraTax CS 12-31-2008"
|
|
→ ("UltraTax CS", "12-31-2008")
|
|
static: "Permanent File"
|
|
→ ("Permanent File",)
|
|
"""
|
|
if folder_type == "yyyy":
|
|
year = match.group(1)
|
|
# Strip "YYYY " from template to get the base folder name
|
|
base_name = template_name.replace("YYYY", "").strip()
|
|
return (base_name, year)
|
|
elif folder_type == "ultratax":
|
|
date = match.group(1)
|
|
return ("UltraTax CS", date)
|
|
else:
|
|
return (template_name,)
|
|
|
|
|
|
def load_lines(path):
|
|
"""Read non-blank, non-comment lines from a file."""
|
|
if not os.path.exists(path):
|
|
return []
|
|
with open(path, "r", encoding="utf-8") as f:
|
|
return [line.strip() for line in f if line.strip() and not line.strip().startswith("#")]
|
|
|
|
|
|
def check_for_clashes(drawer_ids):
|
|
"""Find drawer IDs that are a prefix of another drawer ID.
|
|
|
|
FCCS searches by prefix, so searching the shorter ID (e.g. '02218')
|
|
pops up a selection box when a longer ID exists (e.g. '02218A'),
|
|
which breaks the automated navigation.
|
|
|
|
Returns a list of (short_id, [longer_ids...]) tuples, sorted by short_id.
|
|
Empty list means no clashes.
|
|
"""
|
|
ids = sorted(set(drawer_ids))
|
|
clashes = []
|
|
for short in ids:
|
|
matches = [other for other in ids
|
|
if other != short and other.startswith(short)]
|
|
if matches:
|
|
clashes.append((short, matches))
|
|
return clashes
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# MANIFEST / EXPORT-COMPLETENESS HELPERS
|
|
# ---------------------------------------------------------------------------
|
|
# Shared by fccs_verify.py (batch) and fccs_check.py (interactive) so both
|
|
# judge completeness identically.
|
|
|
|
# Trailing " Page N" (optionally "Page N of M") appended to multi-page exports.
|
|
_PAGE_RE = re.compile(r"\s*Page\s+\d+(?:\s+of\s+\d+)?\s*$", re.IGNORECASE)
|
|
# Creation-date field ("_MM-DD-YYYY_") that precedes the document name.
|
|
_DATE_ANCHOR = re.compile(r"_\d{2}-\d{2}-\d{4}_")
|
|
|
|
|
|
def load_manifest(path):
|
|
"""Load a manifest file and return one row (list of cells) per document.
|
|
|
|
Handles both formats:
|
|
- New: one document per line (tab-separated columns).
|
|
- Old: a raw ListView dump led by 'List1', then row-major cells with
|
|
3 columns per document (Drawer ID, Page Title, Application), reshaped
|
|
so the document count is correct.
|
|
"""
|
|
with open(path, "r", encoding="utf-8") as f:
|
|
lines = [line.rstrip("\n") for line in f if line.strip()]
|
|
|
|
if lines and lines[0].strip() == "List1":
|
|
cells = lines[1:]
|
|
rows = [cells[i:i + 3] for i in range(0, len(cells), 3)]
|
|
return [r for r in rows if len(r) == 3]
|
|
|
|
return [line.split("\t") for line in lines]
|
|
|
|
|
|
def manifest_doc_names(path, drawer_id):
|
|
"""Return the expected document (Page Title) names from a manifest file."""
|
|
rows = load_manifest(path)
|
|
names = []
|
|
for row in rows:
|
|
if len(row) >= 3 and row[0].strip() == drawer_id:
|
|
names.append(row[1]) # old format: DrawerID, PageTitle, Application
|
|
elif row:
|
|
names.append(row[0]) # new format: PageTitle, Application
|
|
return names
|
|
|
|
|
|
def match_key(name):
|
|
"""Normalize a document name for tolerant comparison.
|
|
|
|
Strips a trailing 'Page N' page-split suffix, then reduces to lowercase
|
|
alphanumerics so punctuation and filename-sanitization differences (FCCS
|
|
strips characters illegal in Windows filenames) don't cause false
|
|
mismatches.
|
|
"""
|
|
base = _PAGE_RE.sub("", name)
|
|
return re.sub(r"[^a-z0-9]+", "", base.lower())
|
|
|
|
|
|
def exported_doc_name(filename):
|
|
"""Extract the document-name portion from an exported filename, or None.
|
|
|
|
Format: {drawer}_{client}_{folder}_{MM-DD-YYYY}_{docname}.ext
|
|
The creation-date field is a reliable anchor; the doc name follows the last
|
|
one (client/folder fields don't carry an "_MM-DD-YYYY_" pattern).
|
|
"""
|
|
stem = os.path.splitext(filename)[0]
|
|
anchors = list(_DATE_ANCHOR.finditer(stem))
|
|
if not anchors:
|
|
return None
|
|
return stem[anchors[-1].end():]
|
|
|
|
|
|
def client_name_from_files(files):
|
|
"""Best-effort client name from a drawer's export filenames, or None.
|
|
|
|
Filenames are '{drawer}_{client}_{folder}_{MM-DD-YYYY}_{doc}.ext', so the
|
|
client name is the second underscore-delimited token. Client names carry
|
|
commas/spaces but not underscores; the most common value across the drawer's
|
|
files is returned to shrug off any oddball filename.
|
|
"""
|
|
counts = {}
|
|
for f in files:
|
|
parts = f.split("_")
|
|
if len(parts) >= 2 and parts[1].strip():
|
|
name = parts[1].strip()
|
|
counts[name] = counts.get(name, 0) + 1
|
|
if not counts:
|
|
return None
|
|
return max(counts, key=counts.get)
|
|
|
|
|
|
def index_files_by_drawer(export_dir):
|
|
"""Map each export file to its leading drawer-ID token (before first '_').
|
|
|
|
The underscore boundary keeps clashing IDs separate (04289 vs 04289TS).
|
|
Returns {drawer_id: [filenames]}.
|
|
"""
|
|
index = {}
|
|
for f in os.listdir(export_dir):
|
|
if not os.path.isfile(os.path.join(export_dir, f)):
|
|
continue
|
|
token = f.split("_", 1)[0]
|
|
index.setdefault(token, []).append(f)
|
|
return index
|
|
|
|
|
|
def evaluate_drawer(drawer_id, files, manifest_dir):
|
|
"""Compare a drawer's manifest against its exported files.
|
|
|
|
`files` is the list of export filenames belonging to this drawer. Accounts
|
|
for page-splitting (a document exported as 'Name Page 1/2/...' counts as
|
|
present) and filename sanitization (via match_key).
|
|
|
|
Returns a dict:
|
|
has_manifest, manifest_path, expected (list), missing (list),
|
|
extras (list), unparsed (int), file_count (int)
|
|
"""
|
|
manifest_path = os.path.join(manifest_dir, drawer_id + ".txt")
|
|
result = {
|
|
"has_manifest": os.path.exists(manifest_path),
|
|
"manifest_path": manifest_path,
|
|
"expected": [],
|
|
"missing": [],
|
|
"extras": [],
|
|
"unparsed": 0,
|
|
"file_count": len(files),
|
|
}
|
|
if not result["has_manifest"]:
|
|
return result
|
|
|
|
expected = manifest_doc_names(manifest_path, drawer_id)
|
|
result["expected"] = expected
|
|
|
|
# Group exported files by normalized doc name; page-splits collapse together.
|
|
exported = {} # key -> list of full doc names (one entry per file/page)
|
|
unparsed = 0
|
|
for f in files:
|
|
doc = exported_doc_name(f)
|
|
if doc is None:
|
|
unparsed += 1
|
|
continue
|
|
exported.setdefault(match_key(doc), []).append(doc)
|
|
|
|
result["missing"] = [name for name in expected
|
|
if match_key(name) not in exported]
|
|
|
|
expected_keys = {match_key(n) for n in expected}
|
|
result["extras"] = [names[0] for k, names in exported.items()
|
|
if k not in expected_keys]
|
|
result["unparsed"] = unparsed
|
|
return result
|