# Copyright (C) 2025 James Brotosky, Brandon Wickline # # This program is free software: you can redistribute it and/or modify # it under the terms of the GNU Affero General Public License as published # by the Free Software Foundation, either version 3 of the License, or # (at your option) any later version. # # This program is distributed in the hope that it will be useful, # but WITHOUT ANY WARRANTY; without even the implied warranty of # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the # GNU Affero General Public License for more details. # # You should have received a copy of the GNU Affero General Public License # along with this program. If not, see . import dotenv import gc import json import os import pandas as pd import re import urllib3 import utils.allowlist import utils.getdeviceevents import utils.hashfunctions import utils.pathfunctions import utils.policyfunctions import utils.pretty as ct urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning) dotenv.load_dotenv() #Constants url = os.getenv('url') bad_publisher_list = ["Brave","Zoom", "GlavSoft", "VNC"] pups = ["logmein", "invalid"] badpathparts = ["users", "wwwroot", "windows\\temp", "windows\\task", "windows\\system32", "startup", "windows\\fonts", "Recycle.Bin", "AppData", "programdata"] path_exclusion_constant = 3 min_files_for_path = 4 threat_tolerance_constant = 4 def apivalidation(): match os.getenv('APIKEY'): case '': print(ct.colorText("Please add your API Key to the .env file", "red")) case _: menu_main() def tryToReadCSV(csv): try: df =pd.read_csv(csv) if df.empty: print(ct.colorText("Error: CSV file has headers but no data rows.", "red")) else: print(ct.colorText(f"Data loaded successfully from {csv}", "green")) except pd.errors.EmptyDataError: print(ct.colorText("Notice : CSV file is completely empty (no headers, no data), falling back to empty frame", "white")) df = pd.DataFrame() # Create an empty DataFrame as fallback return df def tryToReadParquet(parquet): try: df = pd.read_parquet(parquet) if df.empty: print(ct.colorText("Error: Parquet file has headers but no data rows.", "red")) else: print(ct.colorText(f"Data loaded successfully from {parquet}", "green")) except pd.errors.EmptyDataError: print(ct.colorText("Notice : Parquet file is completely empty (no headers, no data), falling back to empty frame", "white")) df = pd.DataFrame() # Create an empty DataFrame as fallback return df def deduplicate_list(lst): seen = set() return [x for x in lst if not (x in seen or seen.add(x))] def menu_main(): while True: ct.displayIntro(); print(ct.colorText("1. Get All Events for Single Device", "yellow")) print(ct.colorText("2. Placeholder for Local Approval", "yellow")) print(ct.colorText("3. Placeholder for Another Tool", "yellow")) print(ct.colorText("4. Prepare Policy For Enforcement", "yellow")) print(ct.colorText("Q. Quit", "yellow")) choice = input(ct.colorText("\nEnter Menu Item: ", "white")) if choice == '1': utils.getdeviceevents.devicehistory(url,False) elif choice == "2": menu_local_approve() elif choice == "3": menu_feature2() elif choice == "4": menu_prepare_to_enforce() elif choice == "Q": break else: print(ct.colorText("Invalid choice. Please try again.","red")) def menu_local_approve(): while True: print("\n--- Submenu ---") print("1. Sub-option A") print("2. Sub-option B") print("3. Return to Main Menu") choice = input("Enter your choice: ") if choice == "1": print("You selected Sub-option A") elif choice == "2": print("You selected Sub-option B") elif choice == "3": print("Returning to Main Menu...") break else: print("Invalid choice. Please try again.") def menu_feature2(): while True: print("\n--- Submenu ---") print("1. Sub-option A") print("2. Sub-option B") print("3. Return to Main Menu") choice = input("Enter your choice: ") if choice == "1": print("You selected Sub-option A") elif choice == "2": print("You selected Sub-option B") elif choice == "3": print("Returning to Main Menu...") break else: print("Invalid choice. Please try again.") def menu_prepare_to_enforce(): first_policy = " " second_policy = " " allowlist_parent_name = " " allowlist_child_name = " " destination_name = " " #If the directorys where we're going to store our output dont exist, make them. if not os.path.exists("parquet"): os.makedirs("parquet") if not os.path.exists("needs_approved"): os.makedirs("needs_approved") if not os.path.exists("approved"): os.makedirs("approved") if not os.path.exists("preflight"): os.makedirs("preflight") while True: ct.printEnforceChecklist(first_policy, second_policy, allowlist_child_name, allowlist_parent_name, destination_name) choice = input(ct.colorText("\nEnter your choice: ", "white")) if choice == "1": choice, policynames, policyid = utils.allowlist.listPolicies(url) first_policy = policynames[choice] while True: answer = input(ct.colorText(f"{"Do you want to load a second policy?"} (yes/no): ", "white").strip().lower()) if answer in ("yes", "y"): choice, policynames, policyid = utils.allowlist.listPolicies(url) second_policy = policynames[choice] break elif answer in ("no", "n"): second_policy = first_policy break else: print(ct.colorText("Please answer with 'yes' or 'no'.", "red")) elif choice == "2": if not os.path.exists(f"parquet\\execution_history_{first_policy}.parquet"): print(choice) print(first_policy) exe1 = utils.allowlist.pullPolicyExechistories(url, first_policy, 60, True) data = json.loads(exe1) executionhist_policy1 = pd.DataFrame(data["response"]["exechistories"]) if not executionhist_policy1.empty: executionhist_policy1 = executionhist_policy1[['sha256', 'publisher', 'filename', 'hostname', 'username', 'pprocess', 'gprocess', 'commandline']] executionhist_policy1 = executionhist_policy1.drop_duplicates(subset=['sha256', 'filename', 'hostname']) executionhist_policy1 = executionhist_policy1.sort_values(by=['sha256', 'filename']) executionhist_policy1.to_parquet(f"parquet\\execution_history_{first_policy}.parquet", index=False) print(ct.colorText(f"Staging of Execution history for policy: {first_policy} is complete", "green")) del data del exe1 del executionhist_policy1 gc.collect() if not os.path.exists(f"parquet\\execution_history_{second_policy}.parquet"): exe2 = utils.allowlist.pullPolicyExechistories(url,second_policy, 60, True) data2 = json.loads(exe2) executionhist_policy2 = pd.DataFrame(data2["response"]["exechistories"]) if not executionhist_policy2.empty: executionhist_policy2 = executionhist_policy2[['sha256', 'publisher', 'filename', 'hostname', 'username', 'pprocess', 'gprocess', 'commandline']] executionhist_policy2 = executionhist_policy2.drop_duplicates(subset=['sha256', 'filename', 'hostname']) executionhist_policy2 = executionhist_policy2.sort_values(by=['sha256', 'filename']) executionhist_policy2.to_parquet(f"parquet\\execution_history_{second_policy}.parquet", index=False) print(ct.colorText(f"Staging of Execution history for policy: {second_policy} is complete", "green")) del executionhist_policy2 del data2 del exe2 gc.collect() if not os.path.exists(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"): combined_hashes = pd.DataFrame(columns=['sha256', 'publisher']) hashes = [] try: hash1 = pd.read_parquet(f"parquet\\execution_history_{first_policy}.parquet", columns=['sha256', 'publisher']) utils.pathfunctions.inspect_parquet(f"parquet\\execution_history_{first_policy}.parquet") if not hash1.empty: hashes.append(hash1) else: print("⚠️ First dataframe is empty.") except Exception as e: print(f"❌ Error reading first Parquet file: {e}") try: hash2 = pd.read_parquet(f"parquet\\execution_history_{second_policy}.parquet", columns=['sha256', 'publisher']) utils.pathfunctions.inspect_parquet(f"parquet\\execution_history_{second_policy}.parquet") if not hash2.empty: hashes.append(hash2) else: print("⚠️ Second dataframe is empty.") except Exception as e: print(f"❌ Error reading second Parquet file: {e}") if hashes: combined_hashes = pd.concat(hashes, ignore_index=True) print(f"✅ Combined {len(combined_hashes)} hashes.") else: print("⚠️ No valid dataframes to combine.") combined_hashes = combined_hashes.drop_duplicates(subset=['sha256']) augmented_combo = utils.hashfunctions.augmentAggregatedHashes(url, combined_hashes) numeric_reputation_cols = [ 'reputation_scannermatch', 'reputation_scannercount', 'reputation_threatlevel' ] for col in numeric_reputation_cols: if col in augmented_combo.columns: augmented_combo[col] = pd.to_numeric(augmented_combo[col].replace('N/A', pd.NA), errors='coerce') augmented_combo = augmented_combo.rename(columns={'publisher_x': 'publisher'}) augmented_combo = augmented_combo[['sha256', 'publisher', 'description', 'productname', 'productversion', 'reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount', 'reputation_status', 'reputation_threatlevel', 'reputation_threatname', 'reputation_timestamp']] augmented_combo = augmented_combo.sort_values(by=['publisher', 'description', 'productname']) augmented_combo.to_parquet(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet", index=False) del combined_hashes del augmented_combo gc.collect() print(ct.colorText("Hash reputation info added to dataframe", "green")) if not os.path.exists(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet"): # Categorize the hashes categorized = utils.hashfunctions.categorizeHashes( pd.read_parquet(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"), threat_tolerance_constant, bad_publisher_list, pups ) categorized[0].to_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet", index=False) categorized[1].to_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", index=False) categorized[2].to_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", index=False) del categorized gc.collect() if not os.path.exists(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet"): # Condense execution history try: exe1 = pd.read_parquet(f"parquet\\execution_history_{first_policy}.parquet") utils.pathfunctions.inspect_parquet(f"parquet\\execution_history_{first_policy}.parquet") if not exe1.empty: print() else: print("⚠️ First dataframe is empty.") except Exception as e: print(f"❌ Error reading first Parquet file: {e}") try: exe2 = pd.read_parquet(f"parquet\\execution_history_{second_policy}.parquet") utils.pathfunctions.inspect_parquet(f"parquet\\execution_history_{second_policy}.parquet") if not exe2.empty: print() else: print("⚠️ Second dataframe is empty.") except Exception as e: print(f"❌ Error reading second Parquet file: {e}") if not exe1.empty and not exe2.empty: condensed_combo = pd.concat([exe1, exe2], ignore_index=True) print(f"✅ Combined {len(condensed_combo)} hashes.") elif exe1.empty: condensed_combo = exe2 elif exe2.empty: condensed_combo = exe1 else: print("⚠️ No valid dataframes to combine.") condensed_combo.to_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet", index=False) del condensed_combo gc.collect() if os.path.exists(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") & os.path.exists(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") & os.path.exists(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet"): utils.hashfunctions.combineHashAndHist(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet", first_policy, second_policy) utils.hashfunctions.combineHashAndHist(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", first_policy, second_policy) utils.hashfunctions.combineHashAndHist(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", first_policy, second_policy) unknown = pd.read_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") good = pd.read_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") bad = pd.read_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet") # Build regex pattern once pattern = utils.pathfunctions.regulator(pups) # Move matching rows from unknown and good to bad bad = pd.concat([ bad, unknown[unknown["filename"].str.contains(pattern, na=False)], good[good["filename"].str.contains(pattern, na=False)] ], ignore_index=True) # Remove matching rows from unknown and good unknown = unknown[~unknown["filename"].str.contains(pattern, na=False)] good = good[~good["filename"].str.contains(pattern, na=False)] unknown.to_csv(f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv",index=False) good.to_csv(f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv",index=False) bad.to_csv(f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.csv",index=False) ct.style_dataframe_dark(unknown, f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.html") ct.style_dataframe_dark(good, f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.html") ct.style_dataframe_dark(bad, f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.html") elif choice == "3": if os.path.exists(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv") and os.path.exists(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv"): if not os.path.exists(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet"): df1 = tryToReadCSV(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv") df2 = tryToReadCSV(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv") all_approved_hashes = pd.concat([df1 , df2], ignore_index=True).sort_values(by=['filename']) print(ct.colorText(f"Approved hash lists have been combined","green")) all_approved_hashes.to_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet", index=False) del all_approved_hashes gc.collect() if not os.path.exists(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet"): all_approved_hashes = pd.read_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet") print(ct.colorText(f"Beginning calculating longest common filepaths for path exceptions","green")) haslcp = utils.pathfunctions.split_filepaths_grouped(all_approved_hashes) haslcp.drop_duplicates() forbidden = utils.pathfunctions.regulator(badpathparts, True) forbidden_lcfp = haslcp["longestcfp"].str.contains(forbidden, na=False) print(ct.colorText("Removing forbidden filepaths for path exceptions", "green")) # Make a real DataFrame copy before modifying lcp_not_forbidden = haslcp[~forbidden_lcfp].copy() #For the review, drop down to only the columns we care, and then group by the commmon file path, consolidating and dropping dupes lcp_not_forbidden_review = lcp_not_forbidden[['longestcfp', 'middle', 'filename_only', 'sha256']] # Count unique sha256 per longestcfp unique_sha_counts = lcp_not_forbidden_review.groupby('longestcfp')['sha256'].nunique().reset_index() unique_sha_counts.columns = ['longestcfp', 'unique_sha256_count'] # Merge the count back into the original DataFrame lcp_not_forbidden_review = lcp_not_forbidden_review.merge(unique_sha_counts, on='longestcfp', how='left') lcp_not_forbidden_review = lcp_not_forbidden_review[lcp_not_forbidden_review['unique_sha256_count'] >= min_files_for_path] lcp_not_forbidden_review.to_parquet(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet",index=False) lcp_not_forbidden_review.to_csv(f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.csv",index=False) else: print(ct.colorText(f"Please manually approve hashes prior to this step","red")) elif choice == "4": if os.path.exists(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv"): if not os.path.exists(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet"): allhashes = pd.read_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet") pathexclusions = tryToReadCSV(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv") pathexclusions.to_parquet(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet", index=False) allowbyhash = allhashes[~allhashes['sha256'].isin(pathexclusions['sha256'])] allowbyhash.to_parquet(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet", index=False) easyview = allowbyhash.groupby('sha256').agg(list).reset_index() # Deduplicate all list columns in easyview for col in easyview.columns: if col != 'sha256': # Skip the grouping column easyview[col] = easyview[col].apply(lambda x: list(set(x))) easyview = easyview.sort_values(by=["reputation_status", "filename"]) ct.style_dataframe_dark(easyview, f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html") ct.style_dataframe_dark(pathexclusions, f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html") del allowbyhash del pathexclusions del easyview gc.collect() elif choice == "5": print(ct.colorText(f"Please choose destination_name Policy for Path Exclusions","white")) choice, policynames, policyid = utils.allowlist.listPolicies(url) #print(allowlist_parent_tuple) destination_name = policynames[choice] destination_id = policyid[choice] print(ct.colorText(f"Please choose Parent Allowlist for Known Hashes","white")) choice, allowlists,allowid = utils.allowlist.listAllowlists(url) #print(allowlist_parent_tuple) allowlist_parent_name = allowlists[choice] allowlist_parent_id = allowid[choice] print(ct.colorText(f"Please choose Child Allowlist for Less-Known Hashes","white")) choice, allowlists, allowid = utils.allowlist.listAllowlists(url) #print(allowlist_child_tuple) allowlist_child_name = allowlists[choice] allowlist_child_id = allowid[choice] elif choice == "6": if os.path.exists(f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html") and os.path.exists(f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html") and allowlist_parent_name != " " and allowlist_child_name != " " and destination_name != " ": pathexclusions = pd.read_parquet(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet") allowbyhash = pd.read_parquet(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet") ct.areYouSure() confirmation = input(ct.colorText("Type 'I AGREE' to continue: ","white")) if confirmation.strip().upper() == "I AGREE": print(ct.colorText("Proceeding with the code...", "yellow")) print(ct.colorText(f"Adding path exclusions to {destination_name}", "yellow")) pathexcludelist = pathexclusions['longestcfp'].unique().tolist() # Regex to match a Windows drive letter at the start (e.g., C:\) drive_letter_pattern = re.compile(r'^[a-zA-Z]:\\') # Processed list processed_paths = [ (path if drive_letter_pattern.match(path) else f"\\\\{path}") + "**" for path in pathexcludelist ] utils.policyfunctions.addPath(destination_id,processed_paths) print(ct.colorText(f"Adding hashes to {allowlist_parent_name}", "yellow")) allowlist_parenthashlist = allowbyhash[allowbyhash['reputation_status'] == 'KNOWN']['sha256'].unique().tolist() utils.policyfunctions.addHash(allowlist_parent_id,allowlist_parenthashlist) print(ct.colorText(f"Adding hashes to {allowlist_child_name}", "yellow")) allowlist_childhashlist = allowbyhash[allowbyhash['reputation_status'] == 'UNKNOWN']['sha256'].unique().tolist() utils.policyfunctions.addHash(allowlist_child_id, allowlist_childhashlist) ct.locked() print(repr(processed_paths)) print(processed_paths) exit() else: print(ct.colorText("Operation aborted. You MUST EXPLICITLY AGREE to proceed.", "red")) break elif choice == "Q": break else: print(ct.colorText("Invalid choice. Please try again.", "red")) if __name__ == "__main__": apivalidation()