From b187d8cd79c04d364185a4e366d0432841a3372a Mon Sep 17 00:00:00 2001 From: brotoskyj Date: Mon, 25 Aug 2025 12:09:12 -0400 Subject: [PATCH 1/6] Added colors --- utils/allowlist.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/utils/allowlist.py b/utils/allowlist.py index a3ebf85..0528be7 100644 --- a/utils/allowlist.py +++ b/utils/allowlist.py @@ -17,6 +17,8 @@ import requests import json import os import time +import utils.colortext as ct + def pullPolicyExechistories(url, choice, policiesnames, outputjson: bool): -- 2.34.1 From 5f06e60fddd69a305e5da57f9aef6351420be830 Mon Sep 17 00:00:00 2001 From: = <=> Date: Mon, 25 Aug 2025 12:35:27 -0400 Subject: [PATCH 2/6] New Master candidate --- AirlockTools.py | 44 +++++++++++++++++++------- hashtest.py | 5 --- utils/allowlist.py | 70 +----------------------------------------- utils/hashfunctions.py | 36 ++++++++++++++-------- utils/pathfunctions.py | 30 ++---------------- 5 files changed, 60 insertions(+), 125 deletions(-) delete mode 100644 hashtest.py diff --git a/AirlockTools.py b/AirlockTools.py index 00c5d69..7c1060a 100644 --- a/AirlockTools.py +++ b/AirlockTools.py @@ -19,11 +19,11 @@ import utils.getdeviceevents import utils.allowlist import utils.hashfunctions import utils.pathfunctions +import utils.allowfunctions import utils.colortext as ct import urllib3 import pandas as pd - urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning) dotenv.load_dotenv() @@ -121,7 +121,7 @@ def menu_prepare_to_enforce(): #If the directorys where we're going to store our output dont exist, make them. if not os.path.exists("dataframe_html"): os.makedirs("dataframe_html") if not os.path.exists("dataframe_csv"): os.makedirs("dataframe_csv") - if not os.path.exists("approvals"): os.makedirs("approvals") + if not os.path.exists("manuallyapproved"): os.makedirs("manuallyapproved") df_aggregated_combo = pd.DataFrame() while True: @@ -177,14 +177,25 @@ def menu_prepare_to_enforce(): print(ct.colorText(" [✓] This step has been completed","green")) else: print(ct.colorText(" [✗] This step has not been completed","red")) - + + print(ct.colorText(f"7. Manually review the files \\dataframe_csv\\df_hashes_needing_approval_{first_policy}_{second_policy}.csv and dataframe_csv\\df_automatically_approved_hashes_{first_policy}_{second_policy}.csv", "cyan")) + print(ct.colorText(" Remove the rows containing hashes you do not approve of, and those you would not approve of without metarules.", "cyan")) + print(ct.colorText(" If metarules need to be created, please make note of them, and remove the row from the csv.", "cyan")) + print(ct.colorText(" When complete, move both csv files to the directory 'manuallyapproved' and choose this option to combine these approved hashes with the automatically approved hashes", "cyan")) + + if os.path.isfile(f"dataframe_csv\\df_all_approved_hashes_{first_policy}_{second_policy}.csv"): + print(ct.colorText(" [✓] This step has been completed","green")) + else: + print(ct.colorText(" [✗] This step has not been completed","red")) + + """ print(ct.colorText("7. Compare potential path exclusions with allowed hashes", "cyan")) if os.path.exists(f"dataframe_csv\\df_allowed_paths_{first_policy}_{second_policy}.csv") == True: print(ct.colorText(" [✓] This step has been completed","green")) else: print(ct.colorText(" [✗] This step has not been completed","red")) - + """ print(ct.colorText("Q. Quit", "cyan")) @@ -268,6 +279,21 @@ def menu_prepare_to_enforce(): else: print(ct.colorText(f"Please Augment your data with hash threat info using step 4 prior to attempting this step","red")) + elif choice == "7": + if os.path.isfile(f"manuallyapproved\\df_hashes_needing_approval_{first_policy}_{second_policy}.csv"): + df1 = tryToReadCSV(f"manuallyapproved\\df_hashes_needing_approval_{first_policy}_{second_policy}.csv") + df2 = tryToReadCSV(f"manuallyapproved\\df_automatically_approved_hashes_{first_policy}_{second_policy}.csv") + df_all_approved_hashes = pd.concat([df1 , df2], ignore_index=True) + df_aggregated_combo.to_html(f"dataframe_html\\df_all_approved_hashes_{first_policy}_{second_policy}.html", index=False) + df_aggregated_combo.to_csv(f"dataframe_csv\\df_all_approved_hashes_{first_policy}_{second_policy}.csv", index=False) + + + elif choice == "Q": + break + else: + print(ct.colorText("Invalid choice. Please try again.", "red")) + + """ elif choice == "7": if os.path.exists(f"dataframe_csv\\df_hashes_needing_approval_{first_policy}_{second_policy}.csv") and os.path.exists(f"dataframe_csv\\df_automatically_approved_hashes_{first_policy}_{second_policy}.csv") and os.path.exists(f"dataframe_csv\\df_unapproved_hashes__{first_policy}_{second_policy}.csv"): allowpaths = utils.allowfunctions.filter_and_drop(pd.read_csv(f"dataframe_csv\\df_automatically_approved_hashes_{first_policy}_{second_policy}.csv"),tryToReadCSV(f"dataframe_csv\\df_path_eligible_{first_policy}_{second_policy}.csv"), path_exclusion_constant) @@ -276,13 +302,9 @@ def menu_prepare_to_enforce(): print(ct.colorText(f"Allowable paths determined","green")) else: print(ct.colorText(f"Please complete step 6 prior to attempting this step","red")) - - elif choice == "Q": - break - - else: - print(ct.colorText("Invalid choice. Please try again.", "red")) - + """ + + def tryToReadCSV(csv): try: df =pd.read_csv(csv) diff --git a/hashtest.py b/hashtest.py deleted file mode 100644 index b0dab88..0000000 --- a/hashtest.py +++ /dev/null @@ -1,5 +0,0 @@ -hashes = '' -while True: - inputhash = input("Hash: ") - hashes = hashes + ',' + inputhash - print(hashes) \ No newline at end of file diff --git a/utils/allowlist.py b/utils/allowlist.py index 0528be7..6cc6e2d 100644 --- a/utils/allowlist.py +++ b/utils/allowlist.py @@ -19,7 +19,6 @@ import os import time import utils.colortext as ct - def pullPolicyExechistories(url, choice, policiesnames, outputjson: bool): headers = { @@ -74,71 +73,4 @@ def listPolicies(url): policyids.append(list['groupid']) choice = input(ct.colorText("Select Policy Group: ", "white")) choice = int(choice) - 1 - checkpoint = '000000000000000000000000' - json_output = {'error': 'Success', 'response': {'exechistories': []}} - while True: - json_response_data = checkpoint_stomper(checkpoint, url, policiesnames[choice], headers) - if not json_response_data['response']['exechistories']: - break - for index, item in enumerate(json_response_data['response']['exechistories']): - if index == len(json_response_data['response']['exechistories']) -1: - checkpoint = item['checkpoint'] - print(f"Date Greater than 30 Days, Stepping to new Checkpoint. {item['checkpoint']}") - else: - if (datetime.date.today() - datetime.timedelta(days=10) > datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()): - pass - else: - #json_output['response']['exechistories'].append(json_response_data['response']['exechistories'][1]) - for output in json_response_data['response']['exechistories']: - json_output['response']['exechistories'].append(output) - json_output = json.dumps(json_output) - if outputjson == True: - return json_output - #endpoint = url + '/v1/logging/exechistories' - #payload_dict = { - # "type":[1, 2, 6, 7], - # "checkpoint":"000000000000000000000000", - # "policy": [policiesnames[choice]] - #} - #payload = json.dumps(payload_dict) - #print(payload) - #response = requests.request("POST", endpoint, headers=headers, data=payload, verify=False) - #parse_text = json.loads(response.text) - #text_response = checkpoint_stomper(parse_text['response']['exechistories'], url, policiesnames[choice]) -# - #if outputjson == False: - # return response - # - #parse_text = json.loads(response.text) - # - #for item in parse_text['response']['exechistories']: - # print(item['checkpoint']) - # print(item['datetime']) - # print(item['hostname']) - # print(item['filename']) - # checkpoint_stomper(item['checkpoint'], endpoint, headers, policiesnames[choice]) - -def checkpoint_stomper(checkpoint, url, policy, headers): - endpoint = url + '/v1/logging/exechistories' - payload_dict = { - "type":[1,2,6,7], - "checkpoint": checkpoint, - "policy": [policy] - } - payload = json.dumps(payload_dict) - response = requests.request("POST", endpoint, headers=headers, data=payload, verify=False) - parse_text = json.loads(response.text) - return parse_text - #for index, item in enumerate(parse_text): - # if index == len(parse_text) - 1: - # checkpoint = item['checkpoint'] - # print(f"Time: {item['datetime']} Checkpoint: {item['checkpoint']}") - # response_fuzzer(checkpoint, url, policyname) - # else: - # if (datetime.date.today() - datetime.timedelta(days=30) > datetime.datetime.strptime(item['datetime'].replace( ' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()): - # pass - # else: - # response_fuzzer(checkpoint, url, policyname) - - - print("Finished") + return choice, policiesnames \ No newline at end of file diff --git a/utils/hashfunctions.py b/utils/hashfunctions.py index 9b9c59b..c193284 100644 --- a/utils/hashfunctions.py +++ b/utils/hashfunctions.py @@ -90,28 +90,38 @@ def categorizeHashes(aug_df: pd.DataFrame, threat_tolerance: int, untrusted_publ df = aug_df.copy() - def reputationtool(row, threat_tolerance): - if row["reputation_scannermatch"] == "N/A": - return True + def reputationtool(row): + val = row["reputation_scannermatch"] + if pd.isna(val) or val == "N/A": + return row["publisher_y"] == "Not Signed" try: return int(val) > threat_tolerance except (ValueError, TypeError): return row["publisher_y"] == "Not Signed" - mask_needsreview = (df["publisher_y"] == "Not Signed") & df.apply(lambda row: reputationtool(row, threat_tolerance), axis=1) + df["reputation_flag"] = df.apply(reputationtool, axis=1) - mask_approved = ( - ((df["publisher_y"] != "Not Signed") & (~df["publisher_y"].isin(untrusted_publishers))) & - (df["publisher_y"] != "Not Signed") # explicitly signed - ) | ( - (df["publisher_y"] == "Not Signed") & - (~df.apply(lambda row: reputationtool(row, threat_tolerance), axis=1)) & - (~df["publisher_y"].isin(untrusted_publishers)) # exclude untrusted even if unsigned + mask_needsreview = ( + ((df["publisher_y"] == "Not Signed") & df["reputation_flag"]) | + (df["reputation_status"] == "UNKNOWN") ) + mask_approved = ( + ( + (df["publisher_y"] != "Not Signed") & + ~df["publisher_y"].isin(untrusted_publishers) & + ~df["reputation_status"].isna() + ) | + ( + (df["publisher_y"] == "Not Signed") & + ~df["reputation_flag"] & + ~df["publisher_y"].isin(untrusted_publishers) & + ~df["reputation_status"].isna() + ) + ) needsreview_df = df[mask_needsreview] approved_df = df[mask_approved] - remaining_df = df[~(mask_needsreview | mask_approved)] + unapproved_df = df[~(mask_needsreview | mask_approved)] - return needsreview_df, approved_df, remaining_df + return needsreview_df, approved_df, unapproved_df diff --git a/utils/pathfunctions.py b/utils/pathfunctions.py index 72bdae2..ab8e166 100644 --- a/utils/pathfunctions.py +++ b/utils/pathfunctions.py @@ -57,45 +57,21 @@ def filepathInitialGroup(df: pd.DataFrame): break return join_parts(prefix) - # Step 6: Group directories by shared prefix using custom logic - """ - Loop through each directory path - directories: list of all directory paths. - groups: will hold lists of grouped directories. - used: tracks which directories have already been grouped. - """ + # Step 6: Group directories by shared prefix directories = df["directory"].tolist() groups = [] used = set() - #For Each directory, compare it with others - """ - Skip if already grouped. - Start a new group with the current path. - parts_i is the list of folder names in the path (e.g., ["C:", "Users", "John", "Documents"]). - """ - for i, path in enumerate(directories): if path in used: continue group = [path] parts_i = get_parts(path) - #Compare with all other directories: For each other directory, split it into parts and find the common prefix (shared folder structure). - """ - Logic: - If the directory is deep (>3 parts) and shares at least 3 parts → group it. - If it's exactly 3 parts long and shares at least 2 → group it. - Or, if it shares all but one part and is deep → group it. - These rules are designed to: - Group directories that are closely related in structure. - Avoid grouping unrelated paths that just happen to start similarly. - """ - for j in range(i + 1, len(directories)): parts_j = get_parts(directories[j]) common = os.path.commonprefix([parts_i, parts_j]) - #Apply grouping rules + if (len(parts_i) > 3 and len(common) >= 3) or (len(parts_i) == 3 and len(common) >= 2): group.append(directories[j]) used.add(directories[j]) @@ -123,7 +99,7 @@ def filepathInitialGroup(df: pd.DataFrame): path_eligible = grouped_df[grouped_df["depth"] > 2].drop(columns=["depth"]) path_ineligible = grouped_df[grouped_df["depth"] <= 2].drop(columns=["depth"]) - # Step 10: Move entries from eligible to ineligible if grouped_directory contains 'C:\Users' or 'c$\Users' + # Step 10: Move entries from eligible to ineligible if grouped_directory contains excluded directories mask = path_eligible["grouped_directory"].str.contains(r"(?i)(?:\\Users|\\c\$\\Users|inetpub\\wwwroot|windows\\temp)", na=False) move_to_ineligible = path_eligible[mask] path_eligible = path_eligible[~mask] -- 2.34.1 From d767ba962136bd755279d04ab1fd7d8d4bb88912 Mon Sep 17 00:00:00 2001 From: = <=> Date: Mon, 25 Aug 2025 14:43:37 -0400 Subject: [PATCH 3/6] Everything Ready for the API Calls to approve in airlock --- AirlockTools.py | 94 +++++++++++++++++++++++++++---------------------- 1 file changed, 51 insertions(+), 43 deletions(-) diff --git a/AirlockTools.py b/AirlockTools.py index 7c1060a..20cfaf2 100644 --- a/AirlockTools.py +++ b/AirlockTools.py @@ -19,10 +19,10 @@ import utils.getdeviceevents import utils.allowlist import utils.hashfunctions import utils.pathfunctions -import utils.allowfunctions import utils.colortext as ct import urllib3 import pandas as pd +import ast urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning) dotenv.load_dotenv() @@ -117,6 +117,7 @@ def menu_prepare_to_enforce(): first_policy = " " second_policy = " " + #If the directorys where we're going to store our output dont exist, make them. if not os.path.exists("dataframe_html"): os.makedirs("dataframe_html") @@ -164,38 +165,32 @@ def menu_prepare_to_enforce(): else: print(ct.colorText(" [✗] This step has not been completed","red")) - print(ct.colorText("5. Determine if path exclusions are possible", "cyan")) - - if os.path.exists(f"dataframe_csv\\df_path_eligible_{first_policy}_{second_policy}.csv") == True: - print(ct.colorText(" [✓] This step has been completed","green")) - else: - print(ct.colorText(" [✗] This step has not been completed", "red")) - - print(ct.colorText("6. Categorize your hashes ", "cyan")) + print(ct.colorText("5. Categorize your hashes ", "cyan")) if os.path.isfile(f"dataframe_csv\\df_hashes_needing_approval_{first_policy}_{second_policy}.csv") and os.path.isfile(f"dataframe_csv\\df_automatically_approved_hashes_{first_policy}_{second_policy}.csv") and os.path.isfile(f"dataframe_csv\\df_unapproved_hashes__{first_policy}_{second_policy}.csv"): print(ct.colorText(" [✓] This step has been completed","green")) else: print(ct.colorText(" [✗] This step has not been completed","red")) - print(ct.colorText(f"7. Manually review the files \\dataframe_csv\\df_hashes_needing_approval_{first_policy}_{second_policy}.csv and dataframe_csv\\df_automatically_approved_hashes_{first_policy}_{second_policy}.csv", "cyan")) + print(ct.colorText(f"6. Manually review the files \\dataframe_csv\\df_hashes_needing_approval_{first_policy}_{second_policy}.csv and dataframe_csv\\df_automatically_approved_hashes_{first_policy}_{second_policy}.csv", "cyan")) print(ct.colorText(" Remove the rows containing hashes you do not approve of, and those you would not approve of without metarules.", "cyan")) print(ct.colorText(" If metarules need to be created, please make note of them, and remove the row from the csv.", "cyan")) - print(ct.colorText(" When complete, move both csv files to the directory 'manuallyapproved' and choose this option to combine these approved hashes with the automatically approved hashes", "cyan")) + print(ct.colorText(" When complete, save both csv files to the directory 'manuallyapproved' and choose this option to combine these approved hashes with the automatically approved hashes and generate a list of paths to be reviewed", "cyan")) - if os.path.isfile(f"dataframe_csv\\df_all_approved_hashes_{first_policy}_{second_policy}.csv"): + if os.path.isfile(f"dataframe_csv\\df_paths_needing_review_{first_policy}_{second_policy}.csv"): print(ct.colorText(" [✓] This step has been completed","green")) else: print(ct.colorText(" [✗] This step has not been completed","red")) - """ - print(ct.colorText("7. Compare potential path exclusions with allowed hashes", "cyan")) - - if os.path.exists(f"dataframe_csv\\df_allowed_paths_{first_policy}_{second_policy}.csv") == True: + print(ct.colorText(f"7. Manually review the file df_paths_needing_review_{first_policy}_{second_policy}.csv", "cyan")) + print(ct.colorText(" Remove the rows containing path exclusions you do not approve of" , "cyan")) + print(ct.colorText(" When complete, save the csv file to the directory 'manuallyapproved' and choose this option to generate the proposed list of changes", "cyan")) + + if os.path.isfile(f"manuallyapproved\\df_paths_needing_review_{first_policy}_{second_policy}.csv"): print(ct.colorText(" [✓] This step has been completed","green")) else: print(ct.colorText(" [✗] This step has not been completed","red")) - """ + print(ct.colorText("Q. Quit", "cyan")) @@ -235,11 +230,14 @@ def menu_prepare_to_enforce(): if second_policy is first_policy and os.path.exists(f"dataframe_csv\\df_aggregated_{first_policy}.csv"): df1 = tryToReadCSV(f"dataframe_csv\\df_aggregated_{first_policy}.csv") df_aggregated_combo = df1 + df_aggregated_combo.to_html(f"dataframe_html\\df_aggregated_combo_{first_policy}_{second_policy}.html", index=False) df_aggregated_combo.to_csv(f"dataframe_csv\\df_aggregated_combo_{first_policy}_{second_policy}.csv", index=False) + elif os.path.exists(f"dataframe_csv\\df_aggregated_{first_policy}.csv") and os.path.exists(f"dataframe_csv\\df_aggregated_{second_policy}.csv"): df1 = tryToReadCSV(f"dataframe_csv\\df_aggregated_{first_policy}.csv") df2 = tryToReadCSV(f"dataframe_csv\\df_aggregated_{second_policy}.csv") + df_aggregated_combo = pd.concat([df1 , df2], ignore_index=True) df_aggregated_combo.to_html(f"dataframe_html\\df_aggregated_combo_{first_policy}_{second_policy}.html", index=False) df_aggregated_combo.to_csv(f"dataframe_csv\\df_aggregated_combo_{first_policy}_{second_policy}.csv", index=False) @@ -249,43 +247,63 @@ def menu_prepare_to_enforce(): elif choice == "4": if os.path.exists(f"dataframe_csv\\df_aggregated_combo_{first_policy}_{second_policy}.csv"): df_augmented = utils.hashfunctions.augmentAggregatedHashes(url,tryToReadCSV(f"dataframe_csv\\df_aggregated_combo_{first_policy}_{second_policy}.csv")) + df_augmented.to_html(f"dataframe_html\\df_augmented_combo_{first_policy}_{second_policy}.html", index=False) df_augmented.to_csv(f"dataframe_csv\\df_augmented_combo_{first_policy}_{second_policy}.csv", index=False) + print(ct.colorText(f"Hash reputation info added to dataframe","green")) else: print(ct.colorText(f"Please combine your data with step 3 prior to attempting this step","red")) elif choice == "5": - if os.path.exists(f"dataframe_html\\df_augmented_combo_{first_policy}_{second_policy}.html"): - path_eligible, path_ineligible = utils.pathfunctions.filepathInitialGroup(pd.read_csv(f"dataframe_csv\\df_augmented_combo_{first_policy}_{second_policy}.csv")) - path_eligible.to_html(f"dataframe_html\\df_path_eligible_{first_policy}_{second_policy}.html", index=False) - path_eligible.to_csv(f"dataframe_csv\\df_path_eligible_{first_policy}_{second_policy}.csv", index=False) - path_ineligible.to_html(f"dataframe_html\\df_path_ineligible_{first_policy}_{second_policy}.html", index=False) - path_ineligible.to_csv(f"dataframe_csv\\df_path_ineligible_{first_policy}_{second_policy}.csv", index=False) - print(ct.colorText(f"Eligible paths determined","green")) - else: - print(ct.colorText(f"Please Augment your data with hash threat info using step 4 prior to attempting this step","red")) - - elif choice == "6": if os.path.exists(f"dataframe_csv\\df_augmented_combo_{first_policy}_{second_policy}.csv"): categorized = utils.hashfunctions.categorizeHashes(pd.read_csv(f"dataframe_csv\\df_augmented_combo_{first_policy}_{second_policy}.csv"), threat_tolerance_constant, badpublisherlist) + categorized[0].to_html(f"dataframe_html\\df_hashes_needing_approval_{first_policy}_{second_policy}.html", index=False) categorized[0].to_csv(f"dataframe_csv\\df_hashes_needing_approval_{first_policy}_{second_policy}.csv", index=False) + categorized[1].to_html(f"dataframe_html\\df_automatically_approved_hashes_{first_policy}_{second_policy}.html", index=False) categorized[1].to_csv(f"dataframe_csv\\df_automatically_approved_hashes_{first_policy}_{second_policy}.csv", index=False) + categorized[2].to_html(f"dataframe_html\\df_unapproved_hashes__{first_policy}_{second_policy}.html", index=False) categorized[2].to_csv(f"dataframe_csv\\df_unapproved_hashes__{first_policy}_{second_policy}.csv", index=False) + print(ct.colorText(f"Hashes have been categorized","green")) else: print(ct.colorText(f"Please Augment your data with hash threat info using step 4 prior to attempting this step","red")) - elif choice == "7": - if os.path.isfile(f"manuallyapproved\\df_hashes_needing_approval_{first_policy}_{second_policy}.csv"): + elif choice == "6": + if os.path.exists(f"manuallyapproved\\df_hashes_needing_approval_{first_policy}_{second_policy}.csv") and os.path.exists(f"manuallyapproved\\df_automatically_approved_hashes_{first_policy}_{second_policy}.csv"): df1 = tryToReadCSV(f"manuallyapproved\\df_hashes_needing_approval_{first_policy}_{second_policy}.csv") df2 = tryToReadCSV(f"manuallyapproved\\df_automatically_approved_hashes_{first_policy}_{second_policy}.csv") + df_all_approved_hashes = pd.concat([df1 , df2], ignore_index=True) - df_aggregated_combo.to_html(f"dataframe_html\\df_all_approved_hashes_{first_policy}_{second_policy}.html", index=False) - df_aggregated_combo.to_csv(f"dataframe_csv\\df_all_approved_hashes_{first_policy}_{second_policy}.csv", index=False) + df_all_approved_hashes.to_html(f"dataframe_html\\df_all_approved_hashes_{first_policy}_{second_policy}.html", index=False) + df_all_approved_hashes.to_csv(f"dataframe_csv\\df_all_approved_hashes_{first_policy}_{second_policy}.csv", index=False) + + + df_paths_needing_review, df_path_ineligible = utils.pathfunctions.filepathInitialGroup(pd.read_csv(f"dataframe_csv\\df_all_approved_hashes_{first_policy}_{second_policy}.csv")) + + df_paths_needing_review.to_html(f"dataframe_html\\df_paths_needing_review_{first_policy}_{second_policy}.html", index=False) + df_paths_needing_review.to_csv(f"dataframe_csv\\df_paths_needing_review_{first_policy}_{second_policy}.csv", index=False) + + df_path_ineligible.to_html(f"dataframe_html\\df_path_ineligible_{first_policy}_{second_policy}.html", index=False) + df_path_ineligible.to_csv(f"dataframe_csv\\df_path_ineligible_{first_policy}_{second_policy}.csv", index=False) + + print(ct.colorText(f"Eligible paths determined","green")) + else: + print(ct.colorText(f"Please manually approve hashes prior to this step","red")) + + + elif choice == "7": + if os.path.exists(f"manuallyapproved\\df_paths_needing_review_{first_policy}_{second_policy}.csv"): + df1 = tryToReadCSV(f"manuallyapproved\\df_paths_needing_review_{first_policy}_{second_policy}.csv") + df2 = tryToReadCSV(f"dataframe_csv\\df_path_ineligible_{first_policy}_{second_policy}.csv") + print(df1['grouped_directory']) + print(df2['sha256']) + + + elif choice == "Q": @@ -293,17 +311,7 @@ def menu_prepare_to_enforce(): else: print(ct.colorText("Invalid choice. Please try again.", "red")) - """ - elif choice == "7": - if os.path.exists(f"dataframe_csv\\df_hashes_needing_approval_{first_policy}_{second_policy}.csv") and os.path.exists(f"dataframe_csv\\df_automatically_approved_hashes_{first_policy}_{second_policy}.csv") and os.path.exists(f"dataframe_csv\\df_unapproved_hashes__{first_policy}_{second_policy}.csv"): - allowpaths = utils.allowfunctions.filter_and_drop(pd.read_csv(f"dataframe_csv\\df_automatically_approved_hashes_{first_policy}_{second_policy}.csv"),tryToReadCSV(f"dataframe_csv\\df_path_eligible_{first_policy}_{second_policy}.csv"), path_exclusion_constant) - allowpaths.to_html(f"dataframe_html\\df_allowed_paths_{first_policy}_{second_policy}.html", index=False) - allowpaths.to_csv(f"dataframe_csv\\df_allowed_paths_{first_policy}_{second_policy}.csv", index=False) - print(ct.colorText(f"Allowable paths determined","green")) - else: - print(ct.colorText(f"Please complete step 6 prior to attempting this step","red")) - """ - + def tryToReadCSV(csv): try: -- 2.34.1 From e603ed38d3a85f8472228046945767f53ca2a20d Mon Sep 17 00:00:00 2001 From: brotoskyj Date: Mon, 25 Aug 2025 20:22:51 -0400 Subject: [PATCH 4/6] Fixed Memory Leak Issue \ Refactored Some Sorting for better deduplication --- utils/allowlist.py | 23 +++++++++++++++-------- 1 file changed, 15 insertions(+), 8 deletions(-) diff --git a/utils/allowlist.py b/utils/allowlist.py index 6cc6e2d..f9c6d7a 100644 --- a/utils/allowlist.py +++ b/utils/allowlist.py @@ -18,7 +18,7 @@ import json import os import time import utils.colortext as ct - +import ijson def pullPolicyExechistories(url, choice, policiesnames, outputjson: bool): headers = { @@ -33,18 +33,20 @@ def pullPolicyExechistories(url, choice, policiesnames, outputjson: bool): for index, item in enumerate(json_response_data['response']['exechistories']): if index == len(json_response_data['response']['exechistories']) -1: checkpoint = item['checkpoint'] - print(ct.colorText(f"Date Greater than 30 Days, Stepping to new Checkpoint. {item['checkpoint']}", "blue")) + print(ct.colorText(f"All checkpoints from this execution have been processed. Stepping to the subsequent checkpoint. {item['checkpoint']}", "blue")) + break else: - if (datetime.date.today() - datetime.timedelta(days=10) > datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()): - pass + if (datetime.date.today() - datetime.timedelta(days=30) > datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()): + print(item['datetime']) else: - for output in json_response_data['response']['exechistories']: - json_output['response']['exechistories'].append(output) + print(item['datetime']) + json_output['response']['exechistories'].append(item) json_output = json.dumps(json_output) if outputjson == True: return json_output def checkpoint_stomper(checkpoint, url, policy, headers): + json_output = {'error': 'Success', 'response': {'exechistories': []}} endpoint = url + '/v1/logging/exechistories' payload_dict = { "type":[1,2,6,7], @@ -52,8 +54,13 @@ def checkpoint_stomper(checkpoint, url, policy, headers): "policy": [policy] } payload = json.dumps(payload_dict) - response = requests.request("POST", endpoint, headers=headers, data=payload, verify=False) - parse_text = json.loads(response.text) + with requests.request("POST", endpoint, headers=headers, data=payload, verify=False, stream=True) as response: + parser = ijson.items(response.raw, 'response.exechistories.item') + for item in parser: + key = (item.get('sha256'), item.get('hostname')), item.get('datetime') + if key not in json_output: + json_output['response']['exechistories'].append(item) + parse_text = json.loads(json.dumps(json_output)) return parse_text def listPolicies(url): -- 2.34.1 From f081594a7edf71a43762a72f6e28cb88bcae8fd4 Mon Sep 17 00:00:00 2001 From: brotoskyj Date: Mon, 25 Aug 2025 23:36:24 -0400 Subject: [PATCH 5/6] closes #15 Created iterator to skip over items. This was causing issues because it was too many entries and it would cause a memory leak. --- utils/allowlist.py | 35 ++++++++++++++++++++++++----------- 1 file changed, 24 insertions(+), 11 deletions(-) diff --git a/utils/allowlist.py b/utils/allowlist.py index f9c6d7a..46c48aa 100644 --- a/utils/allowlist.py +++ b/utils/allowlist.py @@ -20,7 +20,6 @@ import time import utils.colortext as ct import ijson def pullPolicyExechistories(url, choice, policiesnames, outputjson: bool): - headers = { "X-APIKey": os.getenv('APIKEY') } @@ -30,17 +29,31 @@ def pullPolicyExechistories(url, choice, policiesnames, outputjson: bool): json_response_data = checkpoint_stomper(checkpoint, url, policiesnames[choice], headers) if not json_response_data['response']['exechistories']: break - for index, item in enumerate(json_response_data['response']['exechistories']): - if index == len(json_response_data['response']['exechistories']) -1: - checkpoint = item['checkpoint'] - print(ct.colorText(f"All checkpoints from this execution have been processed. Stepping to the subsequent checkpoint. {item['checkpoint']}", "blue")) + match_found = False + array_dividend = round(len(json_response_data['response']['exechistories'])/20) + if array_dividend == 0: + array_dividend == 1 + for index, item in enumerate(json_response_data['response']['exechistories'][::array_dividend]): + if (datetime.date.today() - datetime.timedelta(days=30) <= datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()): + print("Found Date Match") + match_found = True break - else: - if (datetime.date.today() - datetime.timedelta(days=30) > datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()): - print(item['datetime']) + checkpoints_processed = round(len(json_response_data['response']['exechistories'])/array_dividend) + print(ct.colorText(f"{checkpoints_processed} checkpoints from this execution have been processed. Stepping to the subsequent checkpoint. {item['checkpoint']}", "blue")) + checkpoint = item['checkpoint'] + if match_found == True: + for index, item in enumerate(json_response_data['response']['exechistories']): + if index == len(json_response_data['response']['exechistories']) -1: + checkpoint = item['checkpoint'] + print(ct.colorText(f"All checkpoints from this execution have been processed. Stepping to the subsequent checkpoint. {item['checkpoint']}", "blue")) + break else: - print(item['datetime']) - json_output['response']['exechistories'].append(item) + if (datetime.date.today() - datetime.timedelta(days=30) > datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()): + pass + else: + print(f"Added: {item['datetime']} | {item['checkpoint']} | {item['filename']} | {item['sha256']}") + json_output['response']['exechistories'].append(item) + match_found = False json_output = json.dumps(json_output) if outputjson == True: return json_output @@ -57,7 +70,7 @@ def checkpoint_stomper(checkpoint, url, policy, headers): with requests.request("POST", endpoint, headers=headers, data=payload, verify=False, stream=True) as response: parser = ijson.items(response.raw, 'response.exechistories.item') for item in parser: - key = (item.get('sha256'), item.get('hostname')), item.get('datetime') + key = (item.get('sha256'), item.get('hostname')) if key not in json_output: json_output['response']['exechistories'].append(item) parse_text = json.loads(json.dumps(json_output)) -- 2.34.1 From ebfb0d80a9d136c3e7793378a09abbd6ad57f2fd Mon Sep 17 00:00:00 2001 From: = <=> Date: Tue, 26 Aug 2025 12:20:09 -0400 Subject: [PATCH 6/6] Pretty HTML and beginnings of Destination Policy --- AirlockTools.py | 47 +++++++------ utils/allowlist.py | 3 +- utils/colortext.py | 14 ---- utils/getdeviceevents.py | 2 +- utils/hashfunctions.py | 16 +++++ utils/pretty.py | 138 +++++++++++++++++++++++++++++++++++++++ 6 files changed, 184 insertions(+), 36 deletions(-) delete mode 100644 utils/colortext.py create mode 100644 utils/pretty.py diff --git a/AirlockTools.py b/AirlockTools.py index 20cfaf2..e0a512f 100644 --- a/AirlockTools.py +++ b/AirlockTools.py @@ -19,7 +19,7 @@ import utils.getdeviceevents import utils.allowlist import utils.hashfunctions import utils.pathfunctions -import utils.colortext as ct +import utils.pretty as ct import urllib3 import pandas as pd import ast @@ -186,7 +186,7 @@ def menu_prepare_to_enforce(): print(ct.colorText(" Remove the rows containing path exclusions you do not approve of" , "cyan")) print(ct.colorText(" When complete, save the csv file to the directory 'manuallyapproved' and choose this option to generate the proposed list of changes", "cyan")) - if os.path.isfile(f"manuallyapproved\\df_paths_needing_review_{first_policy}_{second_policy}.csv"): + if os.path.isfile(f"manuallyapproved\\df_paths_needing_review_{first_policy}_{second_policy}.csv") and os.path.isfile(f"dataframe_csv\\df_hashdestination_{first_policy}_{second_policy}.csv"): print(ct.colorText(" [✓] This step has been completed","green")) else: print(ct.colorText(" [✗] This step has not been completed","red")) @@ -215,43 +215,43 @@ def menu_prepare_to_enforce(): if not os.path.exists("dataframe_csv\\df_aggregated_{first_policy}.csv"): executionhist_policy1 = utils.allowlist.pullPolicyExechistories(url,first_policy_tuple[0], first_policy_tuple[1],True) df_aggregated_policy1 = utils.hashfunctions.aggregateHashes(executionhist_policy1) - df_aggregated_policy1.to_html(f"dataframe_html\\df_aggregated_{first_policy}.html", index=False) df_aggregated_policy1.to_csv(f"dataframe_csv\\df_aggregated_{first_policy}.csv", index=False) + ct.style_dataframe_dark(df_aggregated_policy1, f"dataframe_html\\df_aggregated_{first_policy}.html") print(ct.colorText(f"Staging of Exection history for policy: {first_policy} is complete","green")) if not os.path.exists("dataframe_csv\\df_aggregated_{second_policy}.csv"): executionhist_policy2 = utils.allowlist.pullPolicyExechistories(url,second_policy_tuple[0], second_policy_tuple[1],True) df_aggregated_policy2 = utils.hashfunctions.aggregateHashes(executionhist_policy2) - df_aggregated_policy2.to_html(f"dataframe_html\\df_aggregated_{second_policy}.html", index=False) df_aggregated_policy2.to_csv(f"dataframe_csv\\df_aggregated_{second_policy}.csv", index=False) + ct.style_dataframe_dark(df_aggregated_policy2, f"dataframe_html\\df_aggregated_{second_policy}.html") print(ct.colorText(f"Staging of Exection history for policy: {second_policy} is complete","green")) elif choice == "3": if second_policy is first_policy and os.path.exists(f"dataframe_csv\\df_aggregated_{first_policy}.csv"): df1 = tryToReadCSV(f"dataframe_csv\\df_aggregated_{first_policy}.csv") df_aggregated_combo = df1 - - df_aggregated_combo.to_html(f"dataframe_html\\df_aggregated_combo_{first_policy}_{second_policy}.html", index=False) df_aggregated_combo.to_csv(f"dataframe_csv\\df_aggregated_combo_{first_policy}_{second_policy}.csv", index=False) + ct.style_dataframe_dark(df_aggregated_combo, f"dataframe_html\\df_aggregated_combo_{first_policy}_{second_policy}.html") + print(ct.colorText(f"Dataframes have been aggregated (combined)","green")) elif os.path.exists(f"dataframe_csv\\df_aggregated_{first_policy}.csv") and os.path.exists(f"dataframe_csv\\df_aggregated_{second_policy}.csv"): df1 = tryToReadCSV(f"dataframe_csv\\df_aggregated_{first_policy}.csv") df2 = tryToReadCSV(f"dataframe_csv\\df_aggregated_{second_policy}.csv") - df_aggregated_combo = pd.concat([df1 , df2], ignore_index=True) - df_aggregated_combo.to_html(f"dataframe_html\\df_aggregated_combo_{first_policy}_{second_policy}.html", index=False) df_aggregated_combo.to_csv(f"dataframe_csv\\df_aggregated_combo_{first_policy}_{second_policy}.csv", index=False) + ct.style_dataframe_dark(df_aggregated_combo, f"dataframe_html\\df_aggregated_combo_{first_policy}_{second_policy}.html") + print(ct.colorText(f"Dataframes have been aggregated (combined)","green")) + else: print(ct.colorText(f"Please stage your data before attempting this step","red")) elif choice == "4": if os.path.exists(f"dataframe_csv\\df_aggregated_combo_{first_policy}_{second_policy}.csv"): df_augmented = utils.hashfunctions.augmentAggregatedHashes(url,tryToReadCSV(f"dataframe_csv\\df_aggregated_combo_{first_policy}_{second_policy}.csv")) - - df_augmented.to_html(f"dataframe_html\\df_augmented_combo_{first_policy}_{second_policy}.html", index=False) df_augmented.to_csv(f"dataframe_csv\\df_augmented_combo_{first_policy}_{second_policy}.csv", index=False) - + ct.style_dataframe_dark(df_augmented, f"dataframe_html\\df_augmented_combo_{first_policy}_{second_policy}.html") print(ct.colorText(f"Hash reputation info added to dataframe","green")) + else: print(ct.colorText(f"Please combine your data with step 3 prior to attempting this step","red")) @@ -259,16 +259,17 @@ def menu_prepare_to_enforce(): if os.path.exists(f"dataframe_csv\\df_augmented_combo_{first_policy}_{second_policy}.csv"): categorized = utils.hashfunctions.categorizeHashes(pd.read_csv(f"dataframe_csv\\df_augmented_combo_{first_policy}_{second_policy}.csv"), threat_tolerance_constant, badpublisherlist) - categorized[0].to_html(f"dataframe_html\\df_hashes_needing_approval_{first_policy}_{second_policy}.html", index=False) categorized[0].to_csv(f"dataframe_csv\\df_hashes_needing_approval_{first_policy}_{second_policy}.csv", index=False) + ct.style_dataframe_dark(categorized[0], f"dataframe_html\\df_hashes_needing_approval_{first_policy}_{second_policy}.html") - categorized[1].to_html(f"dataframe_html\\df_automatically_approved_hashes_{first_policy}_{second_policy}.html", index=False) categorized[1].to_csv(f"dataframe_csv\\df_automatically_approved_hashes_{first_policy}_{second_policy}.csv", index=False) + ct.style_dataframe_dark(categorized[1], f"dataframe_html\\df_automatically_approved_hashes_{first_policy}_{second_policy}.html") - categorized[2].to_html(f"dataframe_html\\df_unapproved_hashes__{first_policy}_{second_policy}.html", index=False) categorized[2].to_csv(f"dataframe_csv\\df_unapproved_hashes__{first_policy}_{second_policy}.csv", index=False) + ct.style_dataframe_dark(categorized[2], f"dataframe_html\\df_unapproved_hashes_{first_policy}_{second_policy}.html") print(ct.colorText(f"Hashes have been categorized","green")) + else: print(ct.colorText(f"Please Augment your data with hash threat info using step 4 prior to attempting this step","red")) @@ -278,19 +279,19 @@ def menu_prepare_to_enforce(): df2 = tryToReadCSV(f"manuallyapproved\\df_automatically_approved_hashes_{first_policy}_{second_policy}.csv") df_all_approved_hashes = pd.concat([df1 , df2], ignore_index=True) - df_all_approved_hashes.to_html(f"dataframe_html\\df_all_approved_hashes_{first_policy}_{second_policy}.html", index=False) df_all_approved_hashes.to_csv(f"dataframe_csv\\df_all_approved_hashes_{first_policy}_{second_policy}.csv", index=False) - + ct.style_dataframe_dark(df_all_approved_hashes, f"dataframe_html\\df_all_approved_hashes_{first_policy}_{second_policy}.html") df_paths_needing_review, df_path_ineligible = utils.pathfunctions.filepathInitialGroup(pd.read_csv(f"dataframe_csv\\df_all_approved_hashes_{first_policy}_{second_policy}.csv")) - df_paths_needing_review.to_html(f"dataframe_html\\df_paths_needing_review_{first_policy}_{second_policy}.html", index=False) df_paths_needing_review.to_csv(f"dataframe_csv\\df_paths_needing_review_{first_policy}_{second_policy}.csv", index=False) + ct.style_dataframe_dark(df_paths_needing_review, f"dataframe_html\\df_paths_needing_review_{first_policy}_{second_policy}.html") - df_path_ineligible.to_html(f"dataframe_html\\df_path_ineligible_{first_policy}_{second_policy}.html", index=False) df_path_ineligible.to_csv(f"dataframe_csv\\df_path_ineligible_{first_policy}_{second_policy}.csv", index=False) + ct.style_dataframe_dark(df_path_ineligible, f"dataframe_html\\df_path_ineligible_{first_policy}_{second_policy}.html") print(ct.colorText(f"Eligible paths determined","green")) + else: print(ct.colorText(f"Please manually approve hashes prior to this step","red")) @@ -299,8 +300,14 @@ def menu_prepare_to_enforce(): if os.path.exists(f"manuallyapproved\\df_paths_needing_review_{first_policy}_{second_policy}.csv"): df1 = tryToReadCSV(f"manuallyapproved\\df_paths_needing_review_{first_policy}_{second_policy}.csv") df2 = tryToReadCSV(f"dataframe_csv\\df_path_ineligible_{first_policy}_{second_policy}.csv") - print(df1['grouped_directory']) - print(df2['sha256']) + df3 = tryToReadCSV(f"manuallyapproved\\df_automatically_approved_hashes_{first_policy}_{second_policy}.csv") + df_hashdestination = utils.hashfunctions.destinationbuilder(df2,df3) + df_hashdestination.to_csv("dataframe_csv\\df_hashdestination_{first_policy}_{second_policy}.csv") + ct.style_dataframe_dark(df_hashdestination,f"dataframe_html\\df_hashdestination_{first_policy}_{second_policy}.html") + + else: + print(ct.colorText(f"Please manually approve suggested paths prior to this step","red")) + diff --git a/utils/allowlist.py b/utils/allowlist.py index 46c48aa..c4fb983 100644 --- a/utils/allowlist.py +++ b/utils/allowlist.py @@ -17,8 +17,9 @@ import requests import json import os import time -import utils.colortext as ct +import utils.pretty as ct import ijson + def pullPolicyExechistories(url, choice, policiesnames, outputjson: bool): headers = { "X-APIKey": os.getenv('APIKEY') diff --git a/utils/colortext.py b/utils/colortext.py deleted file mode 100644 index fcd101c..0000000 --- a/utils/colortext.py +++ /dev/null @@ -1,14 +0,0 @@ - -def colorText(text: str, color: str) -> str: - colors = { - "red": "\033[91m", - "green": "\033[92m", - "yellow": "\033[93m", - "blue": "\033[94m", - "magenta": "\033[95m", - "cyan": "\033[96m", - "white": "\033[97m", - "reset": "\033[0m" - } - - return f"{colors.get(color, colors['reset'])}{text}{colors['reset']}" \ No newline at end of file diff --git a/utils/getdeviceevents.py b/utils/getdeviceevents.py index 85bd8dd..fbb5d66 100644 --- a/utils/getdeviceevents.py +++ b/utils/getdeviceevents.py @@ -16,7 +16,7 @@ import datetime import requests import json import os -import utils.colortext as ct +import utils.pretty as ct def devicehistory(url, outputjson: bool): endpoint = url + '/v1/getexechistory' diff --git a/utils/hashfunctions.py b/utils/hashfunctions.py index c193284..771cbe3 100644 --- a/utils/hashfunctions.py +++ b/utils/hashfunctions.py @@ -125,3 +125,19 @@ def categorizeHashes(aug_df: pd.DataFrame, threat_tolerance: int, untrusted_publ unapproved_df = df[~(mask_needsreview | mask_approved)] return needsreview_df, approved_df, unapproved_df + +def destinationbuilder(df, df2): + # Step 1: Explode the 'sha256' list in df2 to create one row per sha256 value + df_expanded = df.explode('sha256') + + # Step 2: Create a new dataframe for the result + df_hashdestination = df_expanded.copy() + + # Step 3: Populate the 'Destination Allowlist' column based on comparison with df3 + df_hashdestination['Destination Allowlist'] = df_hashdestination['sha256'].apply( + lambda x: 'Parent Policy Baseline' if x in df2['sha256'].values else "Destination Policy Allowlist" + ) + + # Step 4: Return the new dataframe + return df_hashdestination + diff --git a/utils/pretty.py b/utils/pretty.py new file mode 100644 index 0000000..58d89d5 --- /dev/null +++ b/utils/pretty.py @@ -0,0 +1,138 @@ + +def colorText(text: str, color: str) -> str: + colors = { + "red": "\033[91m", + "green": "\033[92m", + "yellow": "\033[93m", + "blue": "\033[94m", + "magenta": "\033[95m", + "cyan": "\033[96m", + "white": "\033[97m", + "reset": "\033[0m" + } + + return f"{colors.get(color, colors['reset'])}{text}{colors['reset']}" + +def style_dataframe_dark(df, output_html_path=None, overwrite=True): + from datetime import datetime + + # Get current date and filename for subtitle + today = datetime.now().strftime("%d %B %Y") # Changed to "Day Month Year" + filename = output_html_path.replace('.html', '') if output_html_path else "Report" + + dark_css = """ + + """ + + header = f""" +
+

Airlock Tools

+

{filename} - {today}

+
+ """ + + html_table = df.to_html(index=False, escape=False) + styled_html = ( + f"\n" + f"Airlock Tools Report\n" + f"\n" + f"{dark_css}\n" + f"{header}\n" + f"
\n" + f" {html_table}\n" + f"
\n" + f"\n" + f"" + ) + if output_html_path: + with open(output_html_path, "w", encoding="utf-8") as f: + f.write(styled_html) + print(f"✅ Styled table saved to '{output_html_path}'") + elif overwrite: + import tempfile + temp_path = tempfile.mktemp(suffix=".html") + with open(temp_path, "w", encoding="utf-8") as f: + f.write(styled_html) + print(f"✅ Styled table saved to temporary file: {temp_path}") + else: + return styled_html -- 2.34.1