From 1b300eacb33f4a175c52087170fc8632c724b1a0 Mon Sep 17 00:00:00 2001 From: dat972 Date: Thu, 9 Jul 2026 20:33:07 -0500 Subject: [PATCH] reverted how we compile the ignore list. The longer numbers will actually handle fine so we dont need to ignore. For example 01230 we need to handle but 01230TS we dont. Also we created a separate script to handle clashing numbers specifically as well as helper function that will dump control id's --- README.md | 7 +- __pycache__/fccs_scan.cpython-313.pyc | Bin 6283 -> 6081 bytes fccs_dump_controls.py | 95 ++++++++++++++++++++++++++ fccs_scan.py | 43 ++++++------ 4 files changed, 118 insertions(+), 27 deletions(-) create mode 100644 fccs_dump_controls.py diff --git a/README.md b/README.md index b57ca09..d639f41 100644 --- a/README.md +++ b/README.md @@ -20,6 +20,7 @@ Extracts and organizes documents from FileCabinet CS (Thomson Reuters) using GUI | `fccs_export.py` | Step 2: Automate FCCS GUI to export all drawers | | `fccs_reorganize.py` | Step 3: Parse filenames and rebuild folder structure | | `fccs_verify.py` | Step 4 (optional): Compare manifests against exported files | +| `fccs_dump_controls.py` | Utility: Dump control identifiers of an on-screen dialog | ## Setup Per Engagement @@ -36,12 +37,12 @@ python fccs_scan.py Reads the restored FCCS backup directory and writes all drawer IDs (subfolder names) to `drawer_ids.txt`. It also: -- **Flags prefix clashes** — if one drawer ID is a prefix of another (e.g. `02218` and `02218A`), searching them in FCCS can pop up a selection box that breaks automated navigation. These clashes are reported, and **every drawer in each clash group** is auto-seeded into `ignore.txt` for review. +- **Flags prefix clashes** — if one drawer ID is a prefix of another (e.g. `02218` and `02218A`), searching the **base** ID in FCCS pops up a selection box that breaks plain automated navigation. These clashes are reported, and the base (shorter) IDs are auto-seeded into `ignore.txt`. The longer, more-specific IDs (`02218A`) search fine and export normally; the base IDs are handled by the separate clash-export script. - **Reports ignored drawers** — any IDs listed in `ignore.txt` that exist in this backup are shown as ones the export will skip. -**Ignoring drawers:** The scan creates `ignore.txt` (at `ignore_file`, default `C:\Migration\ignore.txt`) if it doesn't exist and pre-fills it with every drawer in each clash group — searching those in FCCS can show a selection box that stalls the automation, so they're skipped by default. Open the file and: +**Ignoring drawers:** The scan creates `ignore.txt` (at `ignore_file`, default `C:\Migration\ignore.txt`) if it doesn't exist and pre-fills it with the clash base IDs — searching those in FCCS shows a selection box that stalls the plain export, so they're skipped by the main export. Open the file and: -- **Delete or comment out** any clashing drawer you actually want exported (then handle it manually in FCCS). +- **Delete or comment out** any clash base you'd rather handle fully by hand. - **Add** any other drawers to skip (e.g. password-protected folders), one ID per line. Re-running the scan never overwrites your edits — it only appends newly-discovered clashes. `drawer_ids.txt` stays a full inventory; the export skips anything active in the ignore list. Lines starting with `#` are comments. diff --git a/__pycache__/fccs_scan.cpython-313.pyc b/__pycache__/fccs_scan.cpython-313.pyc index ec8de81c1386342933ac8ca53a053060ea59b558..69beba472bc9fc912654691703a232b583774952 100644 GIT binary patch delta 1724 zcmZWqUu+ab7@xhp-P_ySyK?g{0ADxt;4RYqlqz4f-(BwH+NS`;wJme%s1bD ze}41(X3uZCIT&Aw$8`kHwXvh8hOWook)#;^K>D-8FA;pwWCS1Dk8StNfnGeeZ{Gwi zdu7Hy4RtS;9n*F(nJ;_PAHYXkX92H%Qh5>l_2Pb3f;~brRTm69HuH&ve_Og-wRAjp zhS05e)}!PpRwf0zXcxeg)tzG3qqC;tEDVU?$JWwe*ra551ydZmqDkxwt?r6x3~OsY&CL@d;V4mKcE2_aOTJTb9bLdb&o zVFRHPlRH93Lx+Q7*VWc+tQN4*#$)wQYD1(t zOd0$YrKuqT;dicv&uje6cr*Vt+}^GZLn7iyprO#}9D_4qF`ie}REKxh@>`2)LM{{X z5osdqMyHx;ZlIdzgfgM-ltakXpRk4M^7?oS|3^v5V(e5ilEx3HnJthSpcH$I669S) zX)3P(Do}kD6$4O-@z2yvDPt82b2Io1^`HrTs&zGb-r#?$y^_JbhV_ZO(Fed|#AH`> z75-eLv-R{IIfaT+T^nB|s_D*L`Zw0I_S@YR^2$vc-7}{KGCG7Qa<3tIHu{T^ zSjznzTYpcBUog%YmtVW9_5HX3pG%xo>Z)6<=?}vnM?Z>wlD^Zw>sEUAt(MWd+MfFE z#r*eL$9)BLbU)JgTd`boU-qt%eQ)H_{Kd-o{MC`A+-)Pve~)dFUf}DD)Q**5NZ_w7 zqz`=;-Tv>2j^bNbClvB|EH|j$)FL3iY97rE%U^dU##Hs2VSP+qwu~ELsdKqCnUWHN zqH1vsVZ21Me1BtF$}c;OPDmQ!KQ!aT4X{N!gxDyAxF%@35CSsVD1Jf983=`zu)pqNZQ4JYR$$(N7pjS@nl;|%~l(Q6R$XAFMWxR av}FyWF|?F;(6w1gJ{adtTSr*@Uj74@fzJ2< delta 2000 zcmZuyTWk|Y6rHub_WB*$A*sp3<2aO<=0OOF3ZV~_IB5jt5#!J{64Ayxj+d-wt=%f<+*hr&{!R_zx*<)yYCt<<}I1W-HDjArlL zx%bSy_uReKbGq02*z0v67`Gdaoc!sk_Z-8x^iP@JTX{N<4iC1%YMv;pJ%qs+| zvtpq@1Ux;D1(M~9lGcrf69*H=5?EGnM$YF6)v#RDFhE#15X19w5%U?X$V<{Z9zKjq zJe-rc!cJMyz)lB1q&UY@K_V5nd7hS)>9FQ8#s^-5sVeKg_bq0I0CzUi}v~)KBED)Oxe+AFel^x!2&@@Sr0QQ8}&^#*8IHbOR*{Y9;*dW|5Jo~ z01*1e!|>{m{tfG~*7o<<2K|cNssF=96M$=?ZXz%dRBhFe2P-IP9Sx)dAq2QtNK7yt zJYr2cDtMCc!gBaSGyr-*eayZs655Q{i{RJNhElG{rW(R@NPo{BV?z2>d*l94J0usw zq^>NAqE1x1DL|UDa1w#kVsxnViZz0=Oj*}PgQl-6ihGJ#-RT%u?5foN*s$8|awSd? z1&Y5+6`;u}Svm!E41D3MWa*fewOqMu3ku1KbHD_lLMBP3Gep5-!*I%LE?{3KM^r3o zaV)Bs7X%^z7e$&OCVMss95HFDDPn~SWg@FYA{mprcv_y*<6Qr101i_bXeTLz7v?ee zR)LD0F4Ma}x`bi@TmjLkYytp(B}LRUAfHweGelqD+I;@lvwYV-=Y~#HC7TiGP~%vW zu{c9z=mwIgYI||qXINFDDRxMKVUbnCR^YXqVJCBN3}2>}VJ{X0UL%H`hyFnY!*)^@ zsovuZhiozk4Az80468(Fyj~ZbeTz1gfZRSU$(dQzVAULS48u_kn!%QHVlXp?gObve zD1eG2Qlc7;Y8NtWd08L^lQozr!)hupqQRDZ##``OY*UDxZjFSB=A zV>g2R=gHZ?72CykKI}b5mL2QC{(Jt0JD&Ot&U41S;$9uS$u)0qzBAqx@7c(u`Y*$m z!&mxm#P_Us>|Jjh{QCHML*h0!^3aQ#+WrV4F7)xhhCeX4?EKl|TT0w>H*9d;GoBUC z2PbZF?RT5ou5&LeJF2qP<2SjM&HZhzg}`>^?X^z=l7-o) z`|I)6W;jw>%*C{;0PB$wHkaM{SbfuC6t;#NekX;ZHkB16qC`xIt3p3B>q@sNunMA} zVAv_p5s(7y4;hY1kqU;R$_1qi2KxE)sefJRGu3T*UZlz)*d}4B(BaTuYrocaTOby} z#vMD&2BVnuP}VWHsVU0m$<)+k^caMwVuafVHCL3#t4dmz8h5 02218A). + +USAGE +----- +1. In FCCS, get the target screen on-screen (e.g. type the clashing base ID and + click Go so the selection box is showing). +2. From another cmd window (with the venv active), run: + + python fccs_dump_controls.py + + This lists every visible top-level window and writes the full control + identifiers of each to control_dump.txt. + + To narrow it down, filter by title or class: + + python fccs_dump_controls.py --title Select + python fccs_dump_controls.py --class #32770 + +3. Open control_dump.txt, find the selection box, and copy its identifiers + into the references folder (like references/send_to_file_dialog.txt). + +Notes: + - Run from 32-bit Python (same as the rest of the tool). + - Running this from a separate cmd window means the FCCS dialog stays open; + this script does not need focus. +""" + +import argparse +import sys + +from pywinauto import Desktop + + +def main(): + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("--title", default=None, + help="only dump windows whose title contains this text") + ap.add_argument("--class", dest="cls", default=None, + help="only dump windows with this exact class name") + ap.add_argument("--out", default="control_dump.txt", + help="output file (default: control_dump.txt)") + args = ap.parse_args() + + targets = [] + for w in Desktop(backend="win32").windows(): + try: + if not w.is_visible(): + continue + title = w.window_text() + cls = w.class_name() + except Exception: + continue + if args.title and args.title.lower() not in title.lower(): + continue + if args.cls and args.cls != cls: + continue + targets.append(w) + + print(f"Found {len(targets)} matching visible window(s):") + for w in targets: + print(f" handle={w.handle} class={w.class_name()!r} " + f"title={w.window_text()!r}") + + if not targets: + print("Nothing to dump. Is the target window visible? " + "Try without filters, or adjust --title/--class.") + return + + with open(args.out, "w", encoding="utf-8") as f: + old_stdout = sys.stdout + sys.stdout = f + try: + for w in targets: + print("=" * 72) + print(f"WINDOW handle={w.handle} class={w.class_name()!r} " + f"title={w.window_text()!r}") + print("=" * 72) + try: + w.print_control_identifiers() + except Exception as e: + print(f" (could not dump this window: {e})") + print() + finally: + sys.stdout = old_stdout + + print(f"\nControl identifiers written to: {args.out}") + + +if __name__ == "__main__": + main() diff --git a/fccs_scan.py b/fccs_scan.py index 1299714..988231d 100644 --- a/fccs_scan.py +++ b/fccs_scan.py @@ -19,45 +19,40 @@ _IGNORE_HEADER = [ "#", "# Add password-protected or otherwise-excluded drawers here.", "#", - "# The IDs auto-added below are prefix clashes: searching any of them in", - "# FCCS can pop a selection box that breaks the automation, so ALL drawers", - "# in each clash group are skipped by default. DELETE or comment out any", - "# you actually DO want exported (then handle them manually), keep the rest.", + "# The IDs auto-added below are prefix clashes: searching the base ID in", + "# FCCS pops a selection box that breaks the plain export. Only the base", + "# (shorter) ID is listed — the longer, more-specific IDs export normally.", + "# The base IDs are exported separately by fccs_export_clashes.py, which", + "# handles the selection box. DELETE or comment out any you'd rather skip.", "", ] def update_ignore_file(ignore_file, clashes, log): - """Create ignore.txt if missing and seed it with clash-group IDs. + """Create ignore.txt if missing and seed it with clash base (prefix) IDs. - Every drawer involved in a clash (both the prefix and each longer ID that - matches it) is added, since any of them can trip the FCCS selection box. - Never overwrites existing entries — only appends IDs not already present, - and de-dupes so no ID is written twice. Returns the list of IDs added. + Only the shorter (prefix) ID of each clash is added — that's the one that + triggers the FCCS selection box and needs special handling. The longer, + more-specific IDs (e.g. '02218A') search fine and export normally. + Never overwrites existing entries — only appends prefixes not already + present. Returns the list of IDs added. """ existing = set(load_lines(ignore_file)) file_exists = os.path.exists(ignore_file) - seen = set(existing) # never re-add anything already listed - blocks = [] # (comment, [new_ids]) to append, in clash order - for short, matches in clashes: - group = [short] + list(matches) - new_ids = [g for g in group if g not in seen] - if not new_ids: - continue - seen.update(new_ids) - blocks.append((f"# clash group: {', '.join(group)}", new_ids)) + to_add = [(short, matches) for short, matches in clashes + if short not in existing] # Existing file already covers every clash — leave it untouched. - if file_exists and not blocks: + if file_exists and not to_add: return [] lines = [] if not file_exists: lines.extend(_IGNORE_HEADER) - for comment, new_ids in blocks: - lines.append(comment) - lines.extend(new_ids) + for short, matches in to_add: + lines.append(f"# clashes with: {', '.join(matches)}") + lines.append(short) mode = "a" if file_exists else "w" with open(ignore_file, mode, encoding="utf-8") as f: @@ -65,11 +60,11 @@ def update_ignore_file(ignore_file, clashes, log): f.write("\n") # separate the new block from prior content f.write("\n".join(lines) + "\n") - added = [i for _, ids in blocks for i in ids] + added = [short for short, _ in to_add] if not file_exists: log(f"Created ignore file: {ignore_file}") if added: - log(f"Added {len(added)} clash-group ID(s) to ignore list: " + log(f"Added {len(added)} clash base ID(s) to ignore list: " f"{', '.join(added)}") return added