From 935467fbcb9d0bcde3e9d436dc65162964af4f67 Mon Sep 17 00:00:00 2001 From: = <=> Date: Wed, 3 Sep 2025 16:41:27 -0400 Subject: [PATCH] Paths fixed... need to double check the path exclusion format to feed api --- AirlockTools.py | 175 ++++++++++++++++------------------------- utils/hashfunctions.py | 25 +++++- utils/pathfunctions.py | 103 +++++++++++------------- 3 files changed, 133 insertions(+), 170 deletions(-) diff --git a/AirlockTools.py b/AirlockTools.py index dea28dd..e8dc74e 100644 --- a/AirlockTools.py +++ b/AirlockTools.py @@ -18,6 +18,7 @@ import gc import json import os import pandas as pd +import re import urllib3 import utils.allowlist import utils.getdeviceevents @@ -26,13 +27,14 @@ import utils.pathfunctions import utils.policyfunctions import utils.pretty as ct + urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning) dotenv.load_dotenv() #Constants url = os.getenv('url') -badpublisherlist = ["Brave Software, Inc.", "Zoom Video Communications, Inc.", "GlavSoft LLC"] +bad_publisher_list = ["Brave","Zoom", "GlavSoft", "VNC"] pups = ["logmein", "invalid"] badpathparts = ["users", "wwwroot", "windows\\temp", "windows\\task", "windows\\system32", "startup", "windows\\fonts", "Recycle.Bin", "AppData", "programdata"] path_exclusion_constant = 3 @@ -277,7 +279,7 @@ def menu_prepare_to_enforce(): categorized = utils.hashfunctions.categorizeHashes( pd.read_parquet(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"), threat_tolerance_constant, - badpublisherlist, + bad_publisher_list, pups ) @@ -325,86 +327,15 @@ def menu_prepare_to_enforce(): del condensed_combo gc.collect() - if not os.path.exists(f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv"): - - condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet") - needsapproval= pd.read_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") - - #Pull hash info for the entries in the needs approval table - needsapproval = pd.merge(condensed_combo, needsapproval, on='sha256', how='inner') - - #Deduplicate lists in the columns - for col in needsapproval.columns: - if needsapproval[col].apply(lambda x: isinstance(x, list)).all(): - needsapproval[col] = needsapproval[col].apply(deduplicate_list) - - #Rename Publisher, Keep and reorder columns we want - needsapproval = needsapproval.rename(columns={'publisher_x': 'publisher'}) - needsapproval = needsapproval[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']] - - needsapproval['filename_key'] = needsapproval['filename'].apply(lambda x: x[0] if isinstance(x, list) and x else '') - needsapproval = needsapproval.sort_values(by='filename_key').drop(columns=['filename_key']) - - needsapproval.to_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet",index=False) - - del needsapproval - del condensed_combo - gc.collect() - - if not os.path.exists(f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv"): - - condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet") - needsapproval= pd.read_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") - - #Pull hash info for the entries in the needs approval table - needsapproval = pd.merge(condensed_combo, needsapproval, on='sha256', how='inner') - - #Deduplicate lists in the columns - for col in needsapproval.columns: - if needsapproval[col].apply(lambda x: isinstance(x, list)).all(): - needsapproval[col] = needsapproval[col].apply(deduplicate_list) - - #Rename Publisher, Keep and reorder columns we want - needsapproval = needsapproval.rename(columns={'publisher_x': 'publisher'}) - needsapproval = needsapproval[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']] - - needsapproval['filename_key'] = needsapproval['filename'].apply(lambda x: x[0] if isinstance(x, list) and x else '') - needsapproval = needsapproval.sort_values(by='filename_key').drop(columns=['filename_key']) - - needsapproval.to_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", index=False) - - del needsapproval - del condensed_combo - gc.collect() - - if not os.path.exists(f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.html"): - - condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet") - needsapproval= pd.read_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet") - - #Pull hash info for the entries in the needs approval table - needsapproval = pd.merge(condensed_combo, needsapproval, on='sha256', how='inner') - - #Deduplicate lists in the columns - for col in needsapproval.columns: - if needsapproval[col].apply(lambda x: isinstance(x, list)).all(): - needsapproval[col] = needsapproval[col].apply(deduplicate_list) - - #Rename Publisher, Keep and reorder columns we want - needsapproval = needsapproval.rename(columns={'publisher_x': 'publisher'}) - needsapproval = needsapproval[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']] - - needsapproval['filename_key'] = needsapproval['filename'].apply(lambda x: x[0] if isinstance(x, list) and x else '') - needsapproval = needsapproval.sort_values(by='filename_key').drop(columns=['filename_key']) - - needsapproval.to_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", index=False) - - del needsapproval - del condensed_combo - gc.collect() + + if os.path.exists(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") & os.path.exists(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") & os.path.exists(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet"): - + + utils.hashfunctions.combineHashAndHist(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet", first_policy, second_policy) + utils.hashfunctions.combineHashAndHist(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", first_policy, second_policy) + utils.hashfunctions.combineHashAndHist(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", first_policy, second_policy) + unknown = pd.read_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") good = pd.read_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") bad = pd.read_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet") @@ -423,15 +354,14 @@ def menu_prepare_to_enforce(): unknown = unknown[~unknown["filename"].str.contains(pattern, na=False)] good = good[~good["filename"].str.contains(pattern, na=False)] - unknown.to_csv(f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv",index=False) good.to_csv(f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv",index=False) + bad.to_csv(f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.csv",index=False) ct.style_dataframe_dark(unknown, f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.html") ct.style_dataframe_dark(good, f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.html") ct.style_dataframe_dark(bad, f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.html") - elif choice == "3": if os.path.exists(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv") and os.path.exists(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv"): @@ -454,54 +384,67 @@ def menu_prepare_to_enforce(): if not os.path.exists(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet"): all_approved_hashes = pd.read_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet") print(ct.colorText(f"Beginning calculating longest common filepaths for path exceptions","green")) - grouped_df_view, df_with_groups_appended = utils.pathfunctions.export_groups_for_review(all_approved_hashes,"filename","longestcfp",min_files_for_path,path_exclusion_constant) - - df_with_groups_appended.to_parquet(f"parquet\\approved_hashes_with_paths_{first_policy}_{second_policy}.parquet", index=False) - + + haslcp = utils.pathfunctions.split_filepaths_grouped(all_approved_hashes) + haslcp.drop_duplicates() + forbidden = utils.pathfunctions.regulator(badpathparts, True) - forbidden_lcfp = grouped_df_view["longestcfp"].str.contains(forbidden, na=False) - - # Make a real DataFrame copy before modifying - grouped_df_view = grouped_df_view[~forbidden_lcfp].copy() + forbidden_lcfp = haslcp["longestcfp"].str.contains(forbidden, na=False) + print(ct.colorText("Removing forbidden filepaths for path exceptions", "green")) - for col in grouped_df_view.columns: - if grouped_df_view[col].apply(lambda x: isinstance(x, list)).all(): - grouped_df_view[col] = grouped_df_view[col].apply(deduplicate_list) + # Make a real DataFrame copy before modifying + lcp_not_forbidden = haslcp[~forbidden_lcfp].copy() - grouped_df_view.to_parquet(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet", index=False) - grouped_df_view.to_csv(f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.csv", index=False) - ct.style_dataframe_dark(grouped_df_view, f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.html") + #For the review, drop down to only the columns we care, and then group by the commmon file path, consolidating and dropping dupes + lcp_not_forbidden_review = lcp_not_forbidden[['longestcfp', 'middle', 'filename_only', 'sha256']] + + # Count unique sha256 per longestcfp + unique_sha_counts = lcp_not_forbidden_review.groupby('longestcfp')['sha256'].nunique().reset_index() + unique_sha_counts.columns = ['longestcfp', 'unique_sha256_count'] + + # Merge the count back into the original DataFrame + lcp_not_forbidden_review = lcp_not_forbidden_review.merge(unique_sha_counts, on='longestcfp', how='left') + lcp_not_forbidden_review = lcp_not_forbidden_review[lcp_not_forbidden_review['unique_sha256_count'] >= min_files_for_path] + + + + - del grouped_df_view - del df_with_groups_appended - gc.collect() - + lcp_not_forbidden_review.to_parquet(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet",index=False) + lcp_not_forbidden_review.to_csv(f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.csv",index=False) + + else: print(ct.colorText(f"Please manually approve hashes prior to this step","red")) - - + elif choice == "4": - if os.path.exists(f"parquet\\approved_hashes_with_paths_{first_policy}_{second_policy}.parquet") and os.path.exists(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv"): + if os.path.exists(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv"): if not os.path.exists(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet"): - df1 = pd.read_parquet(f"parquet\\approved_hashes_with_paths_{first_policy}_{second_policy}.parquet") - allowbyhash = utils.pathfunctions.mask_from_csv(df1, f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv","longestcfp") + allhashes = pd.read_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet") pathexclusions = tryToReadCSV(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv") - pathexclusions.to_parquet(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet", index=False) + allowbyhash = allhashes[~allhashes['sha256'].isin(pathexclusions['sha256'])] + allowbyhash.to_parquet(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet", index=False) - + easyview = allowbyhash.groupby('sha256').agg(list).reset_index() + # Deduplicate all list columns in easyview + for col in easyview.columns: + if col != 'sha256': # Skip the grouping column + easyview[col] = easyview[col].apply(lambda x: list(set(x))) + + easyview = easyview.sort_values(by=["reputation_status", "filename"]) ct.style_dataframe_dark(easyview, f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html") ct.style_dataframe_dark(pathexclusions, f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html") - del df1 + del allowbyhash del pathexclusions del easyview @@ -542,7 +485,19 @@ def menu_prepare_to_enforce(): print(ct.colorText("Proceeding with the code...", "yellow")) print(ct.colorText(f"Adding path exclusions to {destination_name}", "yellow")) pathexcludelist = pathexclusions['longestcfp'].unique().tolist() - utils.policyfunctions.addPath(destination_id,pathexcludelist) + + + + # Regex to match a Windows drive letter at the start (e.g., C:\) + drive_letter_pattern = re.compile(r'^[a-zA-Z]:\\') + + # Processed list + processed_paths = [ + (path if drive_letter_pattern.match(path) else f"\\\\{path}") + "**" + for path in pathexcludelist +] + + utils.policyfunctions.addPath(destination_id,processed_paths) print(ct.colorText(f"Adding hashes to {allowlist_parent_name}", "yellow")) @@ -554,6 +509,8 @@ def menu_prepare_to_enforce(): utils.policyfunctions.addHash(allowlist_child_id, allowlist_childhashlist) ct.locked() + print(repr(processed_paths)) + print(processed_paths) exit() else: diff --git a/utils/hashfunctions.py b/utils/hashfunctions.py index b869b03..a9e0e17 100644 --- a/utils/hashfunctions.py +++ b/utils/hashfunctions.py @@ -19,6 +19,7 @@ import os import json import utils.pathfunctions as pathf import utils.pretty as ct +import gc def aggregateHashes(executions_json) -> pd.DataFrame: @@ -103,7 +104,7 @@ def augmentAggregatedHashes(url, agg_df: pd.DataFrame) -> pd.DataFrame: return aug_df -def categorizeHashes(df: pd.DataFrame, threat_tolerance: int, untrusted_publishers: list, pups: list): +def categorizeHashes(df: pd.DataFrame, threat_tolerance: int, untrusted_publishers, pups: list): if untrusted_publishers is None: untrusted_publishers = [] if pups is None: pups = [] @@ -126,14 +127,14 @@ def categorizeHashes(df: pd.DataFrame, threat_tolerance: int, untrusted_publishe mask_approved = ( ( (df["publisher"] != "Not Signed") & - ~df["publisher"].isin(untrusted_publishers) & + ~df["publisher"].str.contains(pathf.regulator(untrusted_publishers), case=False, na=False) & ~df["reputation_status"].isna() & ~df["description"].str.contains(pathf.regulator(pups), case=False, na=False) ) | ( (df["publisher"] == "Not Signed") & ~df["reputation_flag"] & - ~df["publisher"].isin(untrusted_publishers) & + ~df["publisher"].str.contains(pathf.regulator(untrusted_publishers), case=False, na=False) & ~df["reputation_status"].isna() & ~df["description"].str.contains(pathf.regulator(pups), case=False, na=False) ) @@ -203,3 +204,21 @@ def destinationHashes( # Concatenate results df_hashdestination = pd.concat([df_paths, df_hashes], ignore_index=True) return df_hashdestination + +def combineHashAndHist(path, first_policy, second_policy): + + condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet") + df = pd.read_parquet(path) + + #Pull hash info for the entries in the needs approval table + df = pd.merge(condensed_combo, df, on='sha256', how='inner') + + #Rename Publisher, Keep and reorder columns we want + df = df.rename(columns={'publisher_x': 'publisher'}) + df = df[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']] + df = df.sort_values(by='filename') + + df.to_parquet(path, index=False) + del df + del condensed_combo + gc.collect() \ No newline at end of file diff --git a/utils/pathfunctions.py b/utils/pathfunctions.py index 2c8fab8..bea88ef 100644 --- a/utils/pathfunctions.py +++ b/utils/pathfunctions.py @@ -15,73 +15,60 @@ import pandas as pd import os -from itertools import chain import ast import re -def split_path(path): - parts = [] - while True: - head, tail = os.path.split(path) - if tail: - parts.insert(0, tail) - path = head - else: - if head: - parts.insert(0, head) - break - return parts +import os +import pandas as pd -def local_common_pass(paths, min_parts=3): - results = {} - paths_sorted = sorted(paths) - for i, path in enumerate(paths_sorted): - candidates = [] +import os +import pandas as pd - if i > 0: - try: - candidates.append(os.path.commonpath([path, paths_sorted[i-1]])) - except ValueError: - # different drives, skip - pass - if i < len(paths_sorted) - 1: - try: - candidates.append(os.path.commonpath([path, paths_sorted[i+1]])) - except ValueError: - # different drives, skip - pass +def split_filepaths_grouped(df, col="filename", group_parts=3, min_parts=3): + def clean_split(path): + parts = os.path.normpath(path).split(os.sep) + # Remove leading empty strings caused by UNC paths + parts = [p for p in parts if p] + return parts - best = path - best_len = 0 - for c in candidates: - parts = split_path(c) - if len(parts) >= min_parts and len(parts) > best_len: - best = c - best_len = len(parts) - results[path] = best - return results + df = df.copy() + split_paths = df[col].apply(clean_split) -def add_longest_common_two_local(df, col="filename_x", new_col="longestcfp", min_parts=3): - dirs_series = df[col].astype(str).apply(os.path.dirname) - first_pass = local_common_pass(dirs_series.tolist(), min_parts) - second_pass = local_common_pass(list(first_pass.values()), min_parts) - df[new_col] = dirs_series.map(lambda d: second_pass[first_pass[d]]) - return df + # Filter out paths with fewer than `min_parts` components + df = df[split_paths.apply(lambda parts: len(parts) >= min_parts)].copy() + split_paths = split_paths[df.index] # Update split_paths to match filtered df -def export_groups_for_review(df, col, group_col, min_number_in_group, path_length_constant): - """ - Compute longest common paths, group filepaths, write CSV for review. - """ - df = df.drop_duplicates(subset=[col], keep='first') - df = add_longest_common_two_local(df, col=col, new_col=group_col) - grouped = df.groupby(group_col)[col].apply(list).reset_index() - grouped = grouped.sort_values(by=col) - print("Before filtering:", len(grouped)) - grouped = grouped[grouped[col].apply(lambda x: len(x) >= min_number_in_group)] - filtered = grouped[grouped[group_col].apply(lambda x: len(os.path.normpath(x).split(os.sep)) >= path_length_constant)] - print("After filtering:", len(grouped)) + df["group_key"] = split_paths.apply(lambda parts: os.sep.join(parts[:group_parts])) + grouped = df.groupby("group_key") + new_rows = [] - return filtered, df + for _, group_df in grouped: + paths = group_df[col].tolist() + split_parts = [clean_split(p) for p in paths] + + def longest_common_prefix(paths): + if not paths: + return [] + prefix = paths[0] + for path in paths[1:]: + prefix = [a for a, b in zip(prefix, path) if a == b] + if not prefix: + break + return prefix + + common_prefix = longest_common_prefix(split_parts) + prefix_str = os.sep.join(common_prefix) + + for i, parts in enumerate(split_parts): + filename = parts[-1] + middle = os.sep.join(parts[len(common_prefix):-1]) if len(parts) > len(common_prefix) + 1 else "" + row = group_df.iloc[i].copy() + row["longestcfp"] = prefix_str + row["middle"] = middle + row["filename_only"] = filename + new_rows.append(row) + + return pd.DataFrame(new_rows).drop(columns=["group_key"]) def mask_from_csv(df, csv_path, filepath_col): """