# Copyright (C) 2025 James Brotosky, Brandon Wickline # # This program is free software: you can redistribute it and/or modify # it under the terms of the GNU Affero General Public License as published # by the Free Software Foundation, either version 3 of the License, or # (at your option) any later version. # # This program is distributed in the hope that it will be useful, # but WITHOUT ANY WARRANTY; without even the implied warranty of # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the # GNU Affero General Public License for more details. # # You should have received a copy of the GNU Affero General Public License # along with this program. If not, see . import dotenv import gc import json import os import pandas as pd import urllib3 import utils.allowlist import utils.getdeviceevents import utils.hashfunctions import utils.pathfunctions import utils.policyfunctions import utils.pretty as ct urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning) dotenv.load_dotenv() #Constants url = os.getenv('url') badpublisherlist = ["Brave Software, Inc.", "Zoom Video Communications, Inc.", "GlavSoft LLC"] pups = ["logmein", "invalid"] badpathparts = ["users", "wwwroot", "windows\\temp", "windows\\task", "windows\\system32", "startup", "windows\\fonts", "Recycle.Bin", "AppData", "programdata"] path_exclusion_constant = 3 min_files_for_path = 4 threat_tolerance_constant = 4 def apivalidation(): match os.getenv('APIKEY'): case '': print(ct.colorText("Please add your API Key to the .env file", "red")) case _: menu_main() def tryToReadCSV(csv): try: df =pd.read_csv(csv) if df.empty: print(ct.colorText("Error: CSV file has headers but no data rows.", "red")) else: print(ct.colorText(f"Data loaded successfully from {csv}", "green")) except pd.errors.EmptyDataError: print(ct.colorText("Notice : CSV file is completely empty (no headers, no data), falling back to empty frame", "white")) df = pd.DataFrame() # Create an empty DataFrame as fallback return df def tryToReadParquet(parquet): try: df = pd.read_parquet(parquet) if df.empty: print(ct.colorText("Error: Parquet file has headers but no data rows.", "red")) else: print(ct.colorText(f"Data loaded successfully from {parquet}", "green")) except pd.errors.EmptyDataError: print(ct.colorText("Notice : Parquet file is completely empty (no headers, no data), falling back to empty frame", "white")) df = pd.DataFrame() # Create an empty DataFrame as fallback return df def deduplicate_list(lst): seen = set() return [x for x in lst if not (x in seen or seen.add(x))] def menu_main(): while True: ct.displayIntro(); print(ct.colorText("1. Get All Events for Single Device", "yellow")) print(ct.colorText("2. Placeholder for Local Approval", "yellow")) print(ct.colorText("3. Placeholder for Another Tool", "yellow")) print(ct.colorText("4. Prepare Policy For Enforcement", "yellow")) print(ct.colorText("Q. Quit", "yellow")) choice = input(ct.colorText("\nEnter Menu Item: ", "white")) if choice == '1': utils.getdeviceevents.devicehistory(url,False) elif choice == "2": menu_local_approve() elif choice == "3": menu_feature2() elif choice == "4": menu_prepare_to_enforce() elif choice == "Q": break else: print(ct.colorText("Invalid choice. Please try again.","red")) def menu_local_approve(): while True: print("\n--- Submenu ---") print("1. Sub-option A") print("2. Sub-option B") print("3. Return to Main Menu") choice = input("Enter your choice: ") if choice == "1": print("You selected Sub-option A") elif choice == "2": print("You selected Sub-option B") elif choice == "3": print("Returning to Main Menu...") break else: print("Invalid choice. Please try again.") def menu_feature2(): while True: print("\n--- Submenu ---") print("1. Sub-option A") print("2. Sub-option B") print("3. Return to Main Menu") choice = input("Enter your choice: ") if choice == "1": print("You selected Sub-option A") elif choice == "2": print("You selected Sub-option B") elif choice == "3": print("Returning to Main Menu...") break else: print("Invalid choice. Please try again.") def menu_prepare_to_enforce(): first_policy = " " second_policy = " " allowlist_parent_name = " " allowlist_child_name = " " destination_name = " " #If the directorys where we're going to store our output dont exist, make them. if not os.path.exists("parquet"): os.makedirs("parquet") if not os.path.exists("needs_approved"): os.makedirs("needs_approved") if not os.path.exists("approved"): os.makedirs("approved") if not os.path.exists("preflight"): os.makedirs("preflight") while True: ct.printEnforceChecklist(first_policy, second_policy, allowlist_child_name, allowlist_parent_name, destination_name) choice = input(ct.colorText("\nEnter your choice: ", "white")) if choice == "1": choice, policynames, policyid = utils.allowlist.listPolicies(url) first_policy = policynames[choice] while True: answer = input(ct.colorText(f"{"Do you want to load a second policy?"} (yes/no): ", "white").strip().lower()) if answer in ("yes", "y"): choice, policynames, policyid = utils.allowlist.listPolicies(url) second_policy = policynames[choice] break elif answer in ("no", "n"): second_policy = first_policy break else: print(ct.colorText("Please answer with 'yes' or 'no'.", "red")) elif choice == "2": if not os.path.exists(f"parquet\\execution_history_{first_policy}.parquet"): print(choice) print(first_policy) exe1 = utils.allowlist.pullPolicyExechistories(url, first_policy, 60, True) data = json.loads(exe1) executionhist_policy1 = pd.DataFrame(data["response"]["exechistories"]) if not executionhist_policy1.empty: executionhist_policy1 = executionhist_policy1[['sha256', 'publisher', 'filename', 'hostname', 'username', 'pprocess', 'gprocess', 'commandline']] executionhist_policy1 = executionhist_policy1.drop_duplicates(subset=['sha256', 'filename', 'hostname']) executionhist_policy1 = executionhist_policy1.sort_values(by=['sha256', 'filename']) executionhist_policy1.to_parquet(f"parquet\\execution_history_{first_policy}.parquet", index=False) print(ct.colorText(f"Staging of Execution history for policy: {first_policy} is complete", "green")) del data del exe1 del executionhist_policy1 gc.collect() if not os.path.exists(f"parquet\\execution_history_{second_policy}.parquet"): exe2 = utils.allowlist.pullPolicyExechistories(url,second_policy, 60, True) data2 = json.loads(exe2) executionhist_policy2 = pd.DataFrame(data2["response"]["exechistories"]) if not executionhist_policy2.empty: executionhist_policy2 = executionhist_policy2[['sha256', 'publisher', 'filename', 'hostname', 'username', 'pprocess', 'gprocess', 'commandline']] executionhist_policy2 = executionhist_policy2.drop_duplicates(subset=['sha256', 'filename', 'hostname']) executionhist_policy2 = executionhist_policy2.sort_values(by=['sha256', 'filename']) executionhist_policy2.to_parquet(f"parquet\\execution_history_{second_policy}.parquet", index=False) print(ct.colorText(f"Staging of Execution history for policy: {second_policy} is complete", "green")) del executionhist_policy2 del data2 del exe2 gc.collect() if not os.path.exists(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"): combined_hashes = pd.DataFrame(columns=['sha256', 'publisher']) hashes = [] try: hash1 = pd.read_parquet(f"parquet\\execution_history_{first_policy}.parquet", columns=['sha256', 'publisher']) utils.pathfunctions.inspect_parquet(f"parquet\\execution_history_{first_policy}.parquet") if not hash1.empty: hashes.append(hash1) else: print("⚠️ First dataframe is empty.") except Exception as e: print(f"❌ Error reading first Parquet file: {e}") try: hash2 = pd.read_parquet(f"parquet\\execution_history_{second_policy}.parquet", columns=['sha256', 'publisher']) utils.pathfunctions.inspect_parquet(f"parquet\\execution_history_{second_policy}.parquet") if not hash2.empty: hashes.append(hash2) else: print("⚠️ Second dataframe is empty.") except Exception as e: print(f"❌ Error reading second Parquet file: {e}") if hashes: combined_hashes = pd.concat(hashes, ignore_index=True) print(f"✅ Combined {len(combined_hashes)} hashes.") else: print("⚠️ No valid dataframes to combine.") combined_hashes = combined_hashes.drop_duplicates(subset=['sha256']) augmented_combo = utils.hashfunctions.augmentAggregatedHashes(url, combined_hashes) numeric_reputation_cols = [ 'reputation_scannermatch', 'reputation_scannercount', 'reputation_threatlevel' ] for col in numeric_reputation_cols: if col in augmented_combo.columns: augmented_combo[col] = pd.to_numeric(augmented_combo[col].replace('N/A', pd.NA), errors='coerce') augmented_combo = augmented_combo.rename(columns={'publisher_x': 'publisher'}) augmented_combo = augmented_combo[['sha256', 'publisher', 'description', 'productname', 'productversion', 'reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount', 'reputation_status', 'reputation_threatlevel', 'reputation_threatname', 'reputation_timestamp']] augmented_combo = augmented_combo.sort_values(by=['publisher', 'description', 'productname']) augmented_combo.to_parquet(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet", index=False) del combined_hashes del augmented_combo gc.collect() print(ct.colorText("Hash reputation info added to dataframe", "green")) if not os.path.exists(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet"): # Categorize the hashes categorized = utils.hashfunctions.categorizeHashes( pd.read_parquet(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"), threat_tolerance_constant, badpublisherlist, pups ) categorized[0].to_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet", index=False) categorized[1].to_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", index=False) categorized[2].to_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", index=False) del categorized gc.collect() if not os.path.exists(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet"): # Condense execution history try: exe1 = pd.read_parquet(f"parquet\\execution_history_{first_policy}.parquet") utils.pathfunctions.inspect_parquet(f"parquet\\execution_history_{first_policy}.parquet") if not exe1.empty: condensed_exe1 = exe1.groupby('sha256').agg(lambda x: list(set(x.tolist()))).reset_index() else: print("⚠️ First dataframe is empty.") except Exception as e: print(f"❌ Error reading first Parquet file: {e}") try: exe2 = pd.read_parquet(f"parquet\\execution_history_{second_policy}.parquet") utils.pathfunctions.inspect_parquet(f"parquet\\execution_history_{second_policy}.parquet") if not exe2.empty: condensed_exe2 = exe2.groupby('sha256').agg(lambda x: list(set(x.tolist()))).reset_index() else: print("⚠️ Second dataframe is empty.") except Exception as e: print(f"❌ Error reading second Parquet file: {e}") if not exe1.empty and not exe2.empty: condensed_combo = pd.concat([condensed_exe1, condensed_exe2], ignore_index=True) condensed_combo = condensed_combo.drop_duplicates().reset_index(drop=True) print(f"✅ Combined {len(condensed_combo)} hashes.") elif exe1.empty: condensed_combo = condensed_exe2 elif exe2.empty: condensed_combo = condensed_exe1 else: print("⚠️ No valid dataframes to combine.") condensed_combo.to_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet", index=False) del condensed_combo gc.collect() if not os.path.exists(f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv"): condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet") needsapproval= pd.read_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") #Pull hash info for the entries in the needs approval table needsapproval = pd.merge(condensed_combo, needsapproval, on='sha256', how='inner') #Deduplicate lists in the columns for col in needsapproval.columns: if needsapproval[col].apply(lambda x: isinstance(x, list)).all(): needsapproval[col] = needsapproval[col].apply(deduplicate_list) #Rename Publisher, Keep and reorder columns we want needsapproval = needsapproval.rename(columns={'publisher_x': 'publisher'}) needsapproval = needsapproval[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']] needsapproval['filename_key'] = needsapproval['filename'].apply(lambda x: x[0] if isinstance(x, list) and x else '') needsapproval = needsapproval.sort_values(by='filename_key').drop(columns=['filename_key']) needsapproval.to_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet",index=False) del needsapproval del condensed_combo gc.collect() if not os.path.exists(f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv"): condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet") needsapproval= pd.read_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") #Pull hash info for the entries in the needs approval table needsapproval = pd.merge(condensed_combo, needsapproval, on='sha256', how='inner') #Deduplicate lists in the columns for col in needsapproval.columns: if needsapproval[col].apply(lambda x: isinstance(x, list)).all(): needsapproval[col] = needsapproval[col].apply(deduplicate_list) #Rename Publisher, Keep and reorder columns we want needsapproval = needsapproval.rename(columns={'publisher_x': 'publisher'}) needsapproval = needsapproval[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']] needsapproval['filename_key'] = needsapproval['filename'].apply(lambda x: x[0] if isinstance(x, list) and x else '') needsapproval = needsapproval.sort_values(by='filename_key').drop(columns=['filename_key']) needsapproval.to_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", index=False) del needsapproval del condensed_combo gc.collect() if not os.path.exists(f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.html"): condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet") needsapproval= pd.read_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet") #Pull hash info for the entries in the needs approval table needsapproval = pd.merge(condensed_combo, needsapproval, on='sha256', how='inner') #Deduplicate lists in the columns for col in needsapproval.columns: if needsapproval[col].apply(lambda x: isinstance(x, list)).all(): needsapproval[col] = needsapproval[col].apply(deduplicate_list) #Rename Publisher, Keep and reorder columns we want needsapproval = needsapproval.rename(columns={'publisher_x': 'publisher'}) needsapproval = needsapproval[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']] needsapproval['filename_key'] = needsapproval['filename'].apply(lambda x: x[0] if isinstance(x, list) and x else '') needsapproval = needsapproval.sort_values(by='filename_key').drop(columns=['filename_key']) needsapproval.to_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", index=False) del needsapproval del condensed_combo gc.collect() if os.path.exists(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") & os.path.exists(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") & os.path.exists(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet"): unknown = pd.read_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") good = pd.read_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") bad = pd.read_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet") # Build regex pattern once pattern = utils.pathfunctions.regulator(pups) # Move matching rows from unknown and good to bad bad = pd.concat([ bad, unknown[unknown["filename"].str.contains(pattern, na=False)], good[good["filename"].str.contains(pattern, na=False)] ], ignore_index=True) # Remove matching rows from unknown and good unknown = unknown[~unknown["filename"].str.contains(pattern, na=False)] good = good[~good["filename"].str.contains(pattern, na=False)] unknown.to_csv(f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv",index=False) good.to_csv(f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv",index=False) ct.style_dataframe_dark(unknown, f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.html") ct.style_dataframe_dark(good, f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.html") ct.style_dataframe_dark(bad, f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.html") elif choice == "3": if os.path.exists(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv") and os.path.exists(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv"): if not os.path.exists(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet"): df1 = tryToReadCSV(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv") df2 = tryToReadCSV(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv") all_approved_hashes = pd.concat([df1 , df2], ignore_index=True).sort_values(by=['filename']) print(ct.colorText(f"Approved hash lists have been combined","green")) all_approved_hashes.to_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet", index=False) del all_approved_hashes gc.collect() if not os.path.exists(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet"): all_approved_hashes = pd.read_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet") print(ct.colorText(f"Beginning calculating longest common filepaths for path exceptions","green")) grouped_df_view, df_with_groups_appended = utils.pathfunctions.export_groups_for_review(all_approved_hashes,"filename","longestcfp",min_files_for_path,path_exclusion_constant) df_with_groups_appended.to_parquet(f"parquet\\approved_hashes_with_paths_{first_policy}_{second_policy}.parquet", index=False) forbidden = utils.pathfunctions.regulator(badpathparts, True) forbidden_lcfp = grouped_df_view["longestcfp"].str.contains(forbidden, na=False) grouped_df_view = grouped_df_view[~forbidden_lcfp] print(ct.colorText(f"Removing forbidden filepaths for path exceptions","green")) grouped_df_view.to_parquet(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet", index=False) grouped_df_view.to_csv(f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.csv", index=False) ct.style_dataframe_dark(grouped_df_view, f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.html") del grouped_df_view del df_with_groups_appended gc.collect() else: print(ct.colorText(f"Please manually approve hashes prior to this step","red")) elif choice == "4": if os.path.exists(f"parquet\\approved_hashes_with_paths_{first_policy}_{second_policy}.parquet") and os.path.exists(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv"): if not os.path.exists(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet"): df1 = pd.read_parquet(f"parquet\\approved_hashes_with_paths_{first_policy}_{second_policy}.parquet") allowbyhash = utils.pathfunctions.mask_from_csv(df1, f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv","longestcfp") pathexclusions = tryToReadCSV(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv") pathexclusions.to_parquet(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet", index=False) allowbyhash.to_parquet(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet", index=False) easyview = allowbyhash.groupby('sha256').agg(list).reset_index() ct.style_dataframe_dark(easyview, f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html") ct.style_dataframe_dark(pathexclusions, f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html") del df1 del allowbyhash del pathexclusions del easyview gc.collect() elif choice == "5": print(ct.colorText(f"Please choose destination_name Policy for Path Exclusions","white")) choice, policynames, policyid = utils.allowlist.listPolicies(url) #print(allowlist_parent_tuple) destination_name = policynames[choice] destination_id = policyid[choice] print(ct.colorText(f"Please choose Parent Allowlist for Known Hashes","white")) choice, allowlists,allowid = utils.allowlist.listAllowlists(url) #print(allowlist_parent_tuple) allowlist_parent_name = allowlists[choice] allowlist_parent_id = allowid[choice] print(ct.colorText(f"Please choose Child Allowlist for Less-Known Hashes","white")) choice, allowlists, allowid = utils.allowlist.listAllowlists(url) #print(allowlist_child_tuple) allowlist_child_name = allowlists[choice] allowlist_child_id = allowid[choice] elif choice == "6": if os.path.exists(f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html") and os.path.exists(f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html") and allowlist_parent_name != " " and allowlist_child_name != " " and destination_name != " ": pathexclusions = pd.read_parquet(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet") allowbyhash = pd.read_parquet(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet") ct.areYouSure() confirmation = input(ct.colorText("Type 'I AGREE' to continue: ","white")) if confirmation.strip().upper() == "I AGREE": print(ct.colorText("Proceeding with the code...", "yellow")) print(ct.colorText(f"Adding path exclusions to {destination_name}", "yellow")) pathexcludelist = pathexclusions['longestcfp'].unique().tolist() utils.policyfunctions.addPath(destination_id,pathexcludelist) print(ct.colorText(f"Adding hashes to {allowlist_parent_name}", "yellow")) allowlist_parenthashlist = allowbyhash[allowbyhash['reputation_status'] == 'KNOWN']['sha256'].unique().tolist() utils.policyfunctions.addHash(allowlist_parent_id,allowlist_parenthashlist) print(ct.colorText(f"Adding hashes to {allowlist_child_name}", "yellow")) allowlist_childhashlist = allowbyhash[allowbyhash['reputation_status'] == 'UNKNOWN']['sha256'].unique().tolist() utils.policyfunctions.addHash(allowlist_child_id, allowlist_childhashlist) ct.locked() exit() else: print(ct.colorText("Operation aborted. You MUST EXPLICITLY AGREE to proceed.", "red")) break elif choice == "Q": break else: print(ct.colorText("Invalid choice. Please try again.", "red")) if __name__ == "__main__": apivalidation()