From 2c1b91fc68e730892f5d4798442d49c1ff2cbfbd Mon Sep 17 00:00:00 2001 From: Driven-Element <129630760+Driven-Element@users.noreply.github.com> Date: Wed, 20 Aug 2025 22:46:55 -0400 Subject: [PATCH] first draft of pathfunctions --- AirlockTools.py | 10 +++++ utils/__pycache__/allowlist.cpython-313.pyc | Bin 2728 -> 2688 bytes .../getdeviceevents.cpython-313.pyc | Bin 3125 -> 3085 bytes .../__pycache__/hashfunctions.cpython-313.pyc | Bin 4753 -> 4708 bytes .../__pycache__/pathfunctions.cpython-313.pyc | Bin 0 -> 2436 bytes utils/hashfunctions.py | 40 ------------------ utils/pathfunctions.py | 39 +++++++++++++++++ 7 files changed, 49 insertions(+), 40 deletions(-) create mode 100644 utils/__pycache__/pathfunctions.cpython-313.pyc create mode 100644 utils/pathfunctions.py diff --git a/AirlockTools.py b/AirlockTools.py index 4bb59a5..db86f5f 100644 --- a/AirlockTools.py +++ b/AirlockTools.py @@ -3,6 +3,7 @@ import os import utils.getdeviceevents import utils.allowlist import utils.hashfunctions +import utils.pathfunctions import urllib3 import pandas as pd @@ -46,6 +47,15 @@ def menu(): categorized[0].to_html("needsreview.html", index=False) categorized[1].to_html("approved.html", index=False) categorized[2].to_html("remaining.html", index=False) + if choice == '4': + html_file = "augmentedlist.html" + augmented_df = pd.read_html(html_file) + print(augmented_df) + + combined_df = pd.concat(augmented_df, ignore_index=True) + test = utils.pathfunctions.filepathInitalGroup(combined_df) + test.to_html("testgroup2.html", index=False) + if __name__ == "__main__": diff --git a/utils/__pycache__/allowlist.cpython-313.pyc b/utils/__pycache__/allowlist.cpython-313.pyc index b9c9af85d3e20277cb6e778e52cbbb050543995e..52a66d4d13e912aee3af23d6c43067ac125577e7 100644 GIT binary patch delta 73 zcmZ1>+91mHnU|M~0SJzzZsa<_YT%S?6%$&VT2vg9k{aWZpIn-onpaXBNR>uMWdVU$8 delta 113 zcmZn=T_MW#nU|M~0SK}UH*%d|jmS#2iU}=FEh>&F&rHtF$;?Ylit*1&bt%d$OI6TS z2oDSORZvs#2zHKf$xklLP0cGQjtMBr&q_@OG8{9Da`Kb2L-O-;;G7Z=N7ryO6I&e% E06nB8!~g&Q diff --git a/utils/__pycache__/getdeviceevents.cpython-313.pyc b/utils/__pycache__/getdeviceevents.cpython-313.pyc index 6613a363b1b63ffab7b2c78c13d534acd02b3ebd..f6c0a32f7567c44095a812451a1b301c350c37b9 100644 GIT binary patch delta 56 zcmdlg(JR6InU|M~0SJzzF5Ad`nnT1R*(xTqIJKxaCM7k-B|o_|H#Kjv0H-)>NPd1! K@n$_vc4h$dOA&Pd delta 114 zcmeB`*eb#OnU|M~0SMAxE#1g{nj@kh*(xTqIJKxaraUt_J0~+QH7UkFFV&?evn*9X zS0Ows)K@`G!6Vo?#w9QaV)4hvetyz&Rx#j;`V6 d66S6`#>7cNQT$&R#28IJ7^yOH6bS+?2LOxvEnxrv diff --git a/utils/__pycache__/pathfunctions.cpython-313.pyc b/utils/__pycache__/pathfunctions.cpython-313.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c645a8650a7bbe1691d061c3d1cfc960e5c1e3d4 GIT binary patch literal 2436 zcmd5-O>7fK6rTNaY$r}q;&{OY>OvqfarqI!6oeugsA(D!T5sG&ikoC(Z_JkUt~&U2}F8oPgQavkt)$+4?WOc@XA6)lUAzKUV2LvAyqx~&DzGqj_9=`?aZ5> z_ulv3d~bHs=kp?H4?h3)a$G>@cRHyxzA9{vfp8B=NMc41W}K15ti&dnxT|)!K1$s7 z>jS*R2hqno<}*xK_>(G(5jH(XTv)O4SWhu^8wju<4!%tni_jD|#ZL)<0DO<2Eboj= zR)CQmVc+GlTnHtQ#8IWV`vogiG2Ys|Ch;e#{-|az97v$JZ<}+RBuK7Ptp0Kj9nG32 zs!MhIRfMlJpH1)yVStrdAV2OGaJehE{!`D>py#c|)79)r;*Ky<>nVoD6ZbdoccFJ) zM@VN9jO0Fj6`?fbe-!dJ!8qC72D{W2L^aG44D~#^AE8zRcft)Yl8<83S@0${dz0K! zyVP-tjqhsSORzPo6l)I(caC8IgTf3EQkP{_C*C<#Q*=42OwSW0G7cMUYDPD(GOg-q zWqu2xD4A$FHZ|x5*?VZ5jf2$K_JOOK+rH|i{3=4@Oq}{|XtZP+TV1~g2-|~J*fD~_ z91(QG(q+QuWa~0%NvoIwU*QNdPguhwY%Xn5jG;n-#8Rh|2~)vlGOZ-f8>xI&(JeFi zmWnkab!FTzw1!FEQelZ&p3CbgOEq-!aBhLP&uDUXHZ2d512j`~e^^&7S-XIZeD3fW zuqJEfAPALXlny41`tJgI6n{MN^F;B34IA3iCa94?-mDNfBm zze$599tcKBSq5JCe`Flfau0KuRx@s=2;H-?=@12X!nBrWOykAfvtiMvKaAYe<%(DGT`+>#0UkUHFpf^ZK_K7Z&EX!1F$ zPE@7mtb;kFeF7b!EMmsOEb*$Qs+*Rqrxe0#s%eE8Oe@l)7DEm3?4cV$=nCk&)8ETK z`pd^-_VL)VTs(Z~(fg0i!QkNE0+kZlfog#;$h-&^3WgH2q1EYLD;u4u2RK>#hoE>% zgCa~5OiiaY_8`0$Ho`m(!9aX-*vL++mV)J34MI+lpoVEZ_VrHe@ToMh6*U``=7M#c^cSxCv!i0Cu{dyDh1wK9QxwIUjph3KJ#)yc*9#m zJF 3: - return True - return False - """ \ No newline at end of file diff --git a/utils/pathfunctions.py b/utils/pathfunctions.py new file mode 100644 index 0000000..ec618e5 --- /dev/null +++ b/utils/pathfunctions.py @@ -0,0 +1,39 @@ +import pandas as pd +import os + + +def filepathInitalGroup(df: pd.DataFrame) -> pd.DataFrame: + import os + import pandas as pd + from itertools import chain + + # 1. Split comma-separated filenames into lists + df["filename_x"] = df["filename_x"].str.split(",") + + # 2. Explode so each filename has its own row + df = df.explode("filename_x", ignore_index=True) + + # 3. Strip whitespace from filenames + df["filename_x"] = df["filename_x"].str.strip() + + # 4. Split into directory and filename + df["directory"] = df["filename_x"].apply(lambda x: os.path.dirname(x) if pd.notna(x) else "") + df["filename"] = df["filename_x"].apply(lambda x: os.path.basename(x) if pd.notna(x) else "") + + # 5. Drop the original column + df = df.drop(columns=["filename_x"]) + + # 6. Ensure all columns are lists (except directory, which is the key) + for col in df.columns: + if col != "directory": + df[col] = df[col].apply(lambda x: x if isinstance(x, list) else [x]) + + # 7. Combine rows with the same directory, flatten lists, deduplicate + def combine_lists(series): + flat = list(chain.from_iterable(series)) + # deduplicate while preserving order + return list(dict.fromkeys(flat)) + + df = df.groupby("directory", as_index=False).agg(combine_lists) + + return df