diff --git a/AirlockTools.py b/AirlockTools.py index a1e7860..ce1f729 100644 --- a/AirlockTools.py +++ b/AirlockTools.py @@ -288,7 +288,7 @@ def menu_prepare_to_enforce(): print(choice) print(first_policy) - exe1 = utils.allowlist.pullPolicyExechistories(url, choice, first_policy, True) + exe1 = utils.allowlist.pullPolicyExechistories(url, first_policy, 60, True) data = json.loads(exe1) executionhist_policy1 = pd.DataFrame(data["response"]["exechistories"]) @@ -299,12 +299,13 @@ def menu_prepare_to_enforce(): executionhist_policy1.to_parquet(f"parquet\\execution_history_{first_policy}.parquet", index=False) print(ct.colorText(f"Staging of Execution history for policy: {first_policy} is complete", "green")) - + + del exe1 del executionhist_policy1 gc.collect() if not os.path.exists(f"parquet\\execution_history_{second_policy}.parquet"): - exe2 = utils.allowlist.pullPolicyExechistories(url, choice, second_policy, True) + exe2 = utils.allowlist.pullPolicyExechistories(url,second_policy, 60, True) data2 = json.loads(exe2) executionhist_policy2 = pd.DataFrame(data2["response"]["exechistories"]) @@ -317,6 +318,7 @@ def menu_prepare_to_enforce(): print(ct.colorText(f"Staging of Execution history for policy: {second_policy} is complete", "green")) del executionhist_policy2 + del exe2 gc.collect() if not os.path.exists(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"): diff --git a/parquet/all_approved_hashes_Developers_Dev Audit.parquet b/parquet/all_approved_hashes_Developers_Dev Audit.parquet deleted file mode 100644 index a78a84a..0000000 Binary files a/parquet/all_approved_hashes_Developers_Dev Audit.parquet and /dev/null differ diff --git a/parquet/approved_hashes_with_paths_Developers_Dev Audit.parquet b/parquet/approved_hashes_with_paths_Developers_Dev Audit.parquet deleted file mode 100644 index 78443c9..0000000 Binary files a/parquet/approved_hashes_with_paths_Developers_Dev Audit.parquet and /dev/null differ diff --git a/parquet/combined_hashlist_Developers_Dev Audit.parquet b/parquet/combined_hashlist_Developers_Dev Audit.parquet deleted file mode 100644 index e99510d..0000000 Binary files a/parquet/combined_hashlist_Developers_Dev Audit.parquet and /dev/null differ diff --git a/parquet/condensed_executions_Developers_Dev Audit.parquet b/parquet/condensed_executions_Developers_Dev Audit.parquet deleted file mode 100644 index 31dd2ea..0000000 Binary files a/parquet/condensed_executions_Developers_Dev Audit.parquet and /dev/null differ diff --git a/parquet/execution_history_Dev Audit.parquet b/parquet/execution_history_Dev Audit.parquet deleted file mode 100644 index f80e99c..0000000 Binary files a/parquet/execution_history_Dev Audit.parquet and /dev/null differ diff --git a/parquet/execution_history_Developers.parquet b/parquet/execution_history_Developers.parquet deleted file mode 100644 index 67f8842..0000000 Binary files a/parquet/execution_history_Developers.parquet and /dev/null differ diff --git a/parquet/hashes_rep_bad_Developers_Dev Audit.parquet b/parquet/hashes_rep_bad_Developers_Dev Audit.parquet deleted file mode 100644 index 88e61d8..0000000 Binary files a/parquet/hashes_rep_bad_Developers_Dev Audit.parquet and /dev/null differ diff --git a/parquet/hashes_rep_good_Developers_Dev Audit.parquet b/parquet/hashes_rep_good_Developers_Dev Audit.parquet deleted file mode 100644 index 5b0d4d8..0000000 Binary files a/parquet/hashes_rep_good_Developers_Dev Audit.parquet and /dev/null differ diff --git a/parquet/hashes_rep_unknown_Developers_Dev Audit.parquet b/parquet/hashes_rep_unknown_Developers_Dev Audit.parquet deleted file mode 100644 index 8e04b75..0000000 Binary files a/parquet/hashes_rep_unknown_Developers_Dev Audit.parquet and /dev/null differ diff --git a/parquet/path_needs_approved_Developers_Dev Audit.parquet b/parquet/path_needs_approved_Developers_Dev Audit.parquet deleted file mode 100644 index 8805207..0000000 Binary files a/parquet/path_needs_approved_Developers_Dev Audit.parquet and /dev/null differ diff --git a/utils/allowlist.py b/utils/allowlist.py index 758ec08..a5aa0dc 100644 --- a/utils/allowlist.py +++ b/utils/allowlist.py @@ -16,48 +16,85 @@ import datetime import requests import json import os +import pandas import time import utils.pretty as ct import ijson +import os +from bson import ObjectId +import datetime + +def pullPolicyExechistories(url, policiesnames, days, outputjson: bool): + file_path = 'chunkinator.json' + + # Initialize file if it doesn't exist + if not os.path.exists(file_path): + with open(file_path, 'w') as file: + json.dump({'error': 'Success', 'response': {'exechistories': []}}, file) + print(f"File '{file_path}' has been created.") + else: + print(f"File '{file_path}' already exists.") + + headers = {"X-APIKey": os.getenv('APIKEY')} -def pullPolicyExechistories(url, choice, policiesnames, outputjson: bool): - headers = { - "X-APIKey": os.getenv('APIKEY') - } checkpoint = '000000000000000000000000' json_output = {'error': 'Success', 'response': {'exechistories': []}} + while True: json_response_data = checkpoint_stomper(checkpoint, url, policiesnames, headers) - if not json_response_data['response']['exechistories']: + histories = json_response_data['response']['exechistories'] + if not histories: break + + array_dividend = max(round(len(histories) / 20), 1) match_found = False - array_dividend = round(len(json_response_data['response']['exechistories'])/20) - if array_dividend == 0: - array_dividend == 1 - for index, item in enumerate(json_response_data['response']['exechistories'][::array_dividend]): - if (datetime.date.today() - datetime.timedelta(days=30) <= datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()): - print("Found Date Match") + + for index, item in enumerate(histories[::array_dividend]): + item_date = datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date() + if item_date >= datetime.date.today() - datetime.timedelta(days): match_found = True break - checkpoints_processed = round(len(json_response_data['response']['exechistories'])/array_dividend) - print(ct.colorText(f"{checkpoints_processed} checkpoints from this execution have been processed. Stepping to the subsequent checkpoint. {item['checkpoint']}", "blue")) + + checkpoints_processed = round(len(histories) / array_dividend) + print(ct.colorText( + f"{index + 1}/{checkpoints_processed} checkpoint(s) processed. " + f"{'Found with Date Match.' if match_found else ''} Last Checkpoint: {item['checkpoint']}.", "blue")) + checkpoint = item['checkpoint'] - if match_found == True: - for index, item in enumerate(json_response_data['response']['exechistories']): - if index == len(json_response_data['response']['exechistories']) -1: - checkpoint = item['checkpoint'] - print(ct.colorText(f"All checkpoints from this execution have been processed. Stepping to the subsequent checkpoint. {item['checkpoint']}", "blue")) - break - else: - if (datetime.date.today() - datetime.timedelta(days=30) > datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()): - pass - else: - print(f"Added: {item['datetime']} | {item['checkpoint']} | {item['filename']} | {item['sha256']}") - json_output['response']['exechistories'].append(item) - match_found = False - json_output = json.dumps(json_output) - if outputjson == True: - return json_output + + if match_found: + for item in histories: + item_date = datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date() + if item_date >= datetime.date.today() - datetime.timedelta(days): + json_output['response']['exechistories'].append(item) + + # Deduplicate and write to file + seen = {} + if os.path.exists(file_path): + with open(file_path, 'r') as file: + existing_data = json.load(file) + combined = existing_data['response']['exechistories'] + json_output['response']['exechistories'] + else: + combined = json_output['response']['exechistories'] + + for item in combined: + key = (item.get('sha256'), item.get('filename'), item.get('hostname')) + seen[key] = item + + deduplicated = list(seen.values()) + with open(file_path, 'w') as file: + json.dump({'error': 'Success', 'response': {'exechistories': deduplicated}}, file) + + # Reset output to free memory + json_output['response']['exechistories'].clear() + + # Final output + with open(file_path, 'r') as file: + final_output = json.load(file) + os.remove(file_path) + + return json.dumps(final_output) if outputjson else None + def checkpoint_stomper(checkpoint, url, policy, headers): json_output = {'error': 'Success', 'response': {'exechistories': []}} @@ -76,6 +113,7 @@ def checkpoint_stomper(checkpoint, url, policy, headers): json_output['response']['exechistories'].append(item) parse_text = json.loads(json.dumps(json_output)) return parse_text + def listPolicies(url): endpoint = url + '/v1/group' print(ct.colorText("[+] Grabbing All Policies", "cyan")) @@ -120,24 +158,4 @@ def listAllowlists(url): #Need else and catch for upper bound -def listCategories(url): - endpoint = url + '/v1/application/categories' - print(ct.colorText("[+] Grabbing All Categories", "cyan")) - payload = {} - headers = { - "X-APIKey": os.getenv('APIKEY') - } - response = requests.request("POST", endpoint, headers=headers, data=payload, verify=False) - parse_text = json.loads(response.text) - policiesnames = [] - policyids = [] - for index, list in enumerate(parse_text['response']['categories'], start=1): - print(ct.colorText(f"{index}. {list['name']}", "yellow")) - policiesnames.append(list['name']) - policyids.append(list['categoryid']) - - - choice = input(ct.colorText("Select Policy Group: ", "white")) - choice = int(choice) - 1 - return choice, policiesnames, policyids