From 2b04210a23b90b51b04334bc73947c116030703d Mon Sep 17 00:00:00 2001 From: dat972 Date: Sat, 11 Jul 2026 12:12:41 -0500 Subject: [PATCH] added a check utility so we can scan the export folder and check if the folder that failed during the export run actually failed or managed to complete succesfully --- README.md | 9 ++ __pycache__/fccs_check.cpython-313.pyc | Bin 0 -> 7758 bytes __pycache__/fccs_config.cpython-313.pyc | Bin 7836 -> 7844 bytes fccs_check.py | 161 ++++++++++++++++++++++++ 4 files changed, 170 insertions(+) create mode 100644 __pycache__/fccs_check.cpython-313.pyc create mode 100644 fccs_check.py diff --git a/README.md b/README.md index 15c6bda..4af0eda 100644 --- a/README.md +++ b/README.md @@ -20,6 +20,7 @@ Extracts and organizes documents from FileCabinet CS (Thomson Reuters) using GUI | `fccs_export.py` | Step 2: Automate FCCS GUI to export all drawers | | `fccs_reorganize.py` | Step 3: Parse filenames and rebuild folder structure | | `fccs_verify.py` | Step 4 (optional): Compare manifests against exported files | +| `fccs_check.py` | Utility: Interactively check if specific drawers exported completely | | `fccs_dump_controls.py` | Utility: Dump control identifiers of an on-screen dialog | ## Setup Per Engagement @@ -68,6 +69,14 @@ python fccs_verify.py During export, each drawer's document list is captured from the FCCS dialog and saved as a manifest. This script compares those manifests against the actual exported files to flag any drawers with missing or extra files. +To spot-check specific drawers (e.g. ones the log marked failed, to see whether they actually finished exporting in the background), run: + +``` +python fccs_check.py +``` + +It prompts for one or more drawer IDs and reports, per drawer, which manifest documents are present vs missing. It accounts for FCCS page-splitting (a document exported as `Name Page 1`, `Name Page 2`, … counts as present) and for filename sanitization (titles containing characters illegal in filenames, like `:`, still match). + ### Step 3: Reorganize Files ``` diff --git a/__pycache__/fccs_check.cpython-313.pyc b/__pycache__/fccs_check.cpython-313.pyc new file mode 100644 index 0000000000000000000000000000000000000000..2fe6db5b88c7c0112595468a3fd3247d39b479cf GIT binary patch literal 7758 zcma)BZ)_XKm7gV-q~w1}VvlGg5^a&PCEHSHC$3`4vK>ouNHga~YKS$tl(wdp z%hmj?ll62eYt^ni&N>w*CjHo#U1vFk!)Vf$2$>?+uVIvQz?-)9 z7W3|~&A3xM-#XHHm}&I!CSC1F@g+H^{Yf{}fg}%gOVR^%a59wi#xEiya0pEX!{~71 zBN$EDrfd;kX%|eAPiazK(q`lM#I%^XjIT_K+O()(wJ2t!oRq;?MYtj=Di$(YNyz7C zaZZvYbz00~@oKT4Xp%e?Q*o}K;DRh-xXct3u^|ng zIjQzw_>N3Fh(c!Ce3OB3f+DKC1fypJS;~p3hP$CJ&K5GI8BvDzQzuR&u_{PeENS>k zp_Dh3>A@MHsFf7ZE?ZJSHJsGMBJPXu!jvG%Ac+XdNqMnpLaL;ZSzt+~Z;fezV)?A7 z^k6~GVnrk~g(31RKO?Cs>1}BPvp)@;h)C^90l!g_l*=kEQ7uRQNipJeyY$}GsCkwKmNd;LAN3f(e>@H?O925b*1qLQ_!cqR(>NU_w9!X}7XPceLgy11kkWU`uBx;*IKH}Yz>i)^>{_f*(xGCg}r^s=5*;#H)PiC+6U5{Gzq3h{P z1J5WtIia@0|C2IQKSUF>kyHuNx~U|Y6fzN0W+qI`Mu`TBHg^Zi1;kD<30HHgac?5y z;l?*QQN&*U&6v3!ur`)6Q=cWMrWNvlc;NU3Vv*$p!zT^b6|klVfUDs5UBn`ItP7JZ zWbAvamU21iY6AfQtV$UI9tCVIHotMLVU>#|nE)_>0SZZ`!$+I<56()toT!L0kq3^| zrSTb|1m2Mo@=ybGQI8eX7_Xx;Kk>Q{Egy(J|GoVYR_D^^!DrHmVco8lF6lN!)E!hG zZ~>OgT4yhT8>+;#80+EQ>B5ZIdqtS-O#nKoy#(ocH#nH?re8~gGb;fYwGaN)C{z{n zl^=QA=8xSxR<%`6);%4SlWQK|{E@egEJPPWix+N03h|mP7HWkw~ zQVb2HKoe5-ty~CX*vW8HI!2?%iRA8vye!OM^z@$@O*Lp5{W?M(gyC!8jBI(h0YeR} z5djt<#5DpU?xsF=f#6yM8!$*AuL<6Wzr?;IW>APNza)l=unJLGw*z+S4wbmB?i7ke zQO-sjI$IR9Y29ta|Fo3VSq0!$cToAdojj1yBnDx{PvnG5BuIm1Y1&VD8eR|^su`#% z=nK!zYFpj2yK-{X>#qzy2(-@G*F63?_3gohH{O1snp_OtdG-CtJCi?p{qAi2nZZ?W zaQ@uQbJf=BD|fbk=EeW&Yne-KZhYiMEo~2Zan6h0A8IF_it&cPBOy^75 zSVY`F*a1U^uL8DY3?^fGK$I98Q#wXqo`gF=-WZff4@w2bAm9K}og(O_#ql*`fv=~p zW%3dPXt*(`58RB7Mo*rMeisTnz_*Bu>oM`F#$SM#mr+EDN71aH5dtCt4TFFt0A0#U z!X@A%f}EKyC`W08gyGaR39$lPAu(OVc|p~95&$W|GZg4GK5qzYz!=avMVTdY3a8hV zN{0bdH6RFbB*s&>9{citphR4{t)MD9z>GRanHH$5vw2C?bZ`3P(1qdj(AbI7;|ZMu zGMkk^f^H{HZzl}~kz%&TN(MSC$t)9UG)U2wp*Utfm?xk zOGm9GT=VR$v3m`lWu;x=6K+%D#qU_!I1V7$2@vi|F#zFXu;;i5 zYk#FUbG+n2`A<$-F5!om<9vL{!;vT+bGzK0KM#y`4o}jRc!|iDFCXnw@HfE zlp@(i0p{o^pX`Nsfekhn6FWen|J8tMZ)gUyEy4O0|MTL1Sekf2lR(l%unqJ9SyXb1 z?3V+7W;MBqZRR&q){4UVx|`Z9{DXGB-(~8LQOaZbi~zKNe#O%?&SUn3ao&Er`9#Kn zb%|Yqbc`<*^rQYJZ!_&OzKy>3H_2?)F|a{qi=|7Xp-W5CxEAy4pi6M$xP~sFhAyG4 zbxFA_`IJ7uerT3&Sq0>Wn&nS9r<|${gD=Q@4r+SmZQR-Afb7qqNNagN;a`fuL-V30 z%A!isPXen3wnIi}09LHb#xw{YWHal`L^=39JT{I;hsMsF8ctrokhqoN@QP;e89|l{ z8vX%f8@XBBeFZW|ZJOjOM#f+wAigw9kz6Yj^2Z}%6orXglL}yhx+*UFEw%!g#!*~u zrlh=fb^E5JnHhHcw#kK7u$eeTw|#iu_xveY|NZ+odSN-_for1LM{ ze6bo`@pMBFz0Y~(UV6txF2aI)?pm#V|J~8M#~&i*=n<%VN4`Rgd*E}<|3#oD556a{7TUG{>y@_bf7Xe|XNm>ha$^x(fNo_|5U^(6V>;eQ(<;*RsMr zwZuL3<1>qyAC298sm}GTa7UK7BlpzLxTB9e$lvA{`0}5 zo|osf54g%m&42E9`}^x{hyH$WQF!mW)i?fT@~4B9(OTP~)vn&3-1y+ea@WAJFY#&D zz@5SB!A}k@9ew5P={a#>=$+}Mu7S#_C12uqtZUBpS0ne~(lK{t?r3d$*T?p|r+?;K z{L6bXAD75R56S=)Lf5%6A7pFgrF7BxvXbkwR%SY@x{hnGq^6s^T zncCjbns;oO8-Ebov02h1khJp;u7EjU@ZIYR1z3pZFj)=!U$7b-5ABF+`TFs*C@{)= z{rk2N;FfZh7c3Xdq6oAmOO=U|0VFfHRFl2$A zOjzUuh>~q%D48Hx4K3exoNUeuBcM)flTKS4GKW#h(f~r(n-ESNO(V9N1>w@0%KtH9 z*QW09c=1`&YNoc<{DG2;o`T@(NIdn;aY<*=VU9FtitJ3WiD$mqGXQ-Sici@$t zLlyvX3Qas`_DT@o-Bf^9_);!&_CWhMi;hBSS(w{kYar7p4#*y8OUa&F=2C9(1@J>4 zu0Vr~a`B@r&pP;ZE2r!eB=@J~aAUuLZ9!8acl0_!9cTii8pVYK*tr<(f!XBH3-&!I z$25oy2V}-SraMT0Qgz;lOC*9R2ViV@?{Fe9o`{=N0ooye6(A=hAvp&p(AhXvNbJ#> zv_V34H=VuHLpqBb(I8X~VN+Nm;e@acZDyoYY50P+p<(jdhRJ*ANhwxEh?S7_W(oaK zivl@sgEK)m3Y-C+)O{&0$d}219{d=fMPlUue{#b)YuPz5srJCJ+ct&3nL_NQJB?#i z-99C1O{~BnsU(a$ituxW24`~+cv+wbn(iXZio9}4Dhd3`ej>#|^Ie7Hb_Q{9NOG~H zDTJ%*-0=9RVMQq@3fbc{{zg1HOOJN#ncS4}5@{y()&rDFq%X~y!TJbkb!dgla8_i8 z+31X<+oh~ZsfS9Ev!-}&M0<7yesfV?g;&?$U!4ahuAnu~wzrQ#7R+y3$lktu>vEkB zS5B-3+ivf;wWGRYv8NvFt(;wL-F|!E)$NKXnQt$|`#^C1%FQeD zH*Vhe@ul~t?o7R(zmvZkukU_-B@kZ<#7WR^3EpF^J0T&pZkuPvF150;xNnKo}0)_XY`F2>t0XhXT*j0Q@gIca5~6 zUxoOQPWD&W3H7hr_>m6#uQ3BP91N$)I;PW+R^64RCmw0TKdBR)CVt^D4u0wH0!=P$ zG#e>|an$G{r{&KkkT7(KF$S{v=yBPM#5ybEhkoy7R{~h)I20aU|;q}kQ--_3^ z$3J=LpGW_3wC*`u;XViFD8E&%wI8Sl4pzL6ogCBuc)ObkKHlkJIv+nBVLBdPWWr4B a;aLW;-nnw!zUw~A{q-qeHEcW8=zjrkOA&>RAWc{UTH F3;=_#4-^0Z delta 43 xcmZ2tJI9vqGcPX}0}wnF2*?r^-N;wZ#3!9>72}+rk{aXY>>M1kc`*~C3;+U03*Z0% diff --git a/fccs_check.py b/fccs_check.py new file mode 100644 index 0000000..b5bb608 --- /dev/null +++ b/fccs_check.py @@ -0,0 +1,161 @@ +""" +Utility: Check whether specific drawers actually finished exporting. + +Interactive — prompts for one or more drawer IDs, then for each drawer compares +its manifest (the documents FCCS said it would export, captured during Step 2) +against the files actually sitting in the export folder, and reports any +missing documents. + +Handles two quirks of the FCCS export: + + 1. Page-splitting: a single manifest document (e.g. "Donations") is exported + as one file if it's a single page, or as "Donations Page 1", + "Donations Page 2", ... for a multi-page document. All of those count as + that one document being present. + + 2. Filename sanitization: FCCS strips characters that are illegal in Windows + filenames (: / \\ ? * " < > |) from document titles, so a manifest title + like 'US Tax Return (... 01:35PM)' won't match the exported file + character-for-character. Comparison is done on a normalized key + (lowercase, alphanumerics only) so these still match. + +USAGE +----- + python fccs_check.py + Drawer ID(s): 08097 18430 +""" + +import os +import re +import sys + +from fccs_config import parse_args, load_config +from fccs_verify import load_manifest # reuse the both-format manifest loader + + +# Trailing " Page N" (optionally "Page N of M") appended to multi-page exports. +_PAGE_RE = re.compile(r"\s*Page\s+\d+(?:\s+of\s+\d+)?\s*$", re.IGNORECASE) +# Creation-date field ("_MM-DD-YYYY_") that precedes the document name. +_DATE_ANCHOR = re.compile(r"_\d{2}-\d{2}-\d{4}_") + + +def match_key(name): + """Normalize a document name for tolerant comparison. + + Strips a trailing 'Page N' page-split suffix, then reduces to lowercase + alphanumerics so punctuation and filename-sanitization differences don't + cause false mismatches. + """ + base = _PAGE_RE.sub("", name) + return re.sub(r"[^a-z0-9]+", "", base.lower()) + + +def manifest_doc_names(path, drawer_id): + """Return the expected document (Page Title) names from a manifest file.""" + rows = load_manifest(path) + names = [] + for row in rows: + if len(row) >= 3 and row[0].strip() == drawer_id: + names.append(row[1]) # old format: DrawerID, PageTitle, Application + elif row: + names.append(row[0]) # new format: PageTitle, Application + return names + + +def exported_doc_name(filename): + """Extract the document-name portion from an exported filename, or None. + + Format: {drawer}_{client}_{folder}_{MM-DD-YYYY}_{docname}.ext + The creation-date field is a reliable anchor; the doc name follows the last + one (the client/folder fields don't carry an "_MM-DD-YYYY_" pattern). + """ + stem = os.path.splitext(filename)[0] + anchors = list(_DATE_ANCHOR.finditer(stem)) + if not anchors: + return None + return stem[anchors[-1].end():] + + +def check_drawer(drawer_id, files, manifest_dir, out): + """Report completeness of one drawer's export.""" + manifest_path = os.path.join(manifest_dir, drawer_id + ".txt") + if not os.path.exists(manifest_path): + out("") + out(f"[{drawer_id}] NO MANIFEST at {manifest_path} — cannot verify " + "(was this drawer exported by the tool?)") + return + + expected = manifest_doc_names(manifest_path, drawer_id) + + # Group exported files by normalized doc name; page-splits collapse together. + exported = {} # key -> list of full doc names (one entry per file/page) + unparsed = [] + for f in files: + doc = exported_doc_name(f) + if doc is None: + unparsed.append(f) + continue + exported.setdefault(match_key(doc), []).append(doc) + + missing = [name for name in expected if match_key(name) not in exported] + + expected_keys = {match_key(n) for n in expected} + extras = [names[0] for k, names in exported.items() if k not in expected_keys] + + out("") + out(f"[{drawer_id}] manifest lists {len(expected)} document(s); " + f"{len(files)} file(s) in export folder.") + if missing: + out(f" INCOMPLETE — {len(missing)} document(s) missing from export:") + for m in missing: + out(f" - {m}") + else: + out(f" COMPLETE — all {len(expected)} manifest document(s) present.") + if extras: + out(f" Note: {len(extras)} exported document(s) not in the manifest:") + for e in extras: + out(f" - {e}") + if unparsed: + out(f" Note: {len(unparsed)} file(s) had no recognizable date anchor " + "and were skipped.") + + +def main(): + args = parse_args() + cfg = load_config(args.config) + export_dir = cfg.get("paths", "export_dir") + manifest_dir = cfg.get("paths", "manifest_dir") + + if not os.path.isdir(export_dir): + print(f"ERROR: export directory not found: {export_dir}") + sys.exit(1) + + # Index every export file by its leading drawer-ID token (before first "_"). + # The underscore boundary keeps clashing IDs separate (04289 vs 04289TS). + files_by_drawer = {} + for f in os.listdir(export_dir): + if not os.path.isfile(os.path.join(export_dir, f)): + continue + token = f.split("_", 1)[0] + files_by_drawer.setdefault(token, []).append(f) + + print("FCCS export completeness check") + print(f" export folder : {export_dir}") + print(f" manifests : {manifest_dir}") + print("Enter drawer ID(s) separated by spaces or commas (blank to quit).") + + while True: + try: + raw = input("\nDrawer ID(s): ").strip() + except EOFError: + break + if not raw: + break + ids = [i for i in re.split(r"[\s,]+", raw) if i] + for drawer_id in ids: + check_drawer(drawer_id, files_by_drawer.get(drawer_id, []), + manifest_dir, print) + + +if __name__ == "__main__": + main()