From 2f47ee7e44b2c97b9679a1fd7b92668e0609bb5a Mon Sep 17 00:00:00 2001 From: brotoskyj Date: Wed, 3 Sep 2025 11:31:59 -0400 Subject: [PATCH] Quality of Life updates --- requirements.txt | 1 + utils/allowlist.py | 78 ++++++++++++++++++++++++---------------------- 2 files changed, 41 insertions(+), 38 deletions(-) diff --git a/requirements.txt b/requirements.txt index 3ba3d18..015b684 100644 --- a/requirements.txt +++ b/requirements.txt @@ -2,3 +2,4 @@ pandas==2.3.2 python-dotenv==1.1.1 Requests==2.32.5 urllib3==2.5.0 +tqdm \ No newline at end of file diff --git a/utils/allowlist.py b/utils/allowlist.py index 4f9af7f..2d0f621 100644 --- a/utils/allowlist.py +++ b/utils/allowlist.py @@ -21,6 +21,8 @@ import ijson import os from bson import ObjectId import datetime +import tqdm +import time def pullPolicyExechistories(url, policiesnames, days, outputjson: bool): file_path = 'chunkinator.json' @@ -33,46 +35,46 @@ def pullPolicyExechistories(url, policiesnames, days, outputjson: bool): headers = {"X-APIKey": os.getenv('APIKEY')} checkpoint = str(skipback(days)) json_output = {'error': 'Success', 'response': {'exechistories': []}} - while True: - json_response_data = checkpoint_stomper(checkpoint, url, policiesnames, headers) - histories = json_response_data['response']['exechistories'] - if not histories: - break - array_dividend = max(round(len(histories) / 20), 1) - match_found = False - for index, item in enumerate(histories[::array_dividend]): - if (datetime.date.today() - datetime.timedelta(days=days) <= datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()): - match_found = True + with tqdm.tqdm(total=100, desc="Total Percentage Complete: ") as pbar: + while True: + json_response_data = checkpoint_stomper(checkpoint, url, policiesnames, headers) + histories = json_response_data['response']['exechistories'] + if not histories: break - checkpoints_processed = round(len(histories) / array_dividend) - if ( index + 1 ) < checkpoints_processed: - print(ct.colorText(f"{index + 1}/{checkpoints_processed} checkpoint(s) from this execution have been processed with date match. Last Checkpoint: {checkpoint}", "blue")) - else: - print(ct.colorText(f"{index + 1}/{checkpoints_processed} checkpoint(s) Processed. Last Checkpoint: {checkpoint}", "blue")) - if match_found == True: - for index, item in enumerate(histories): - if index == len(histories) - 1: - print(ct.colorText(f"All Events Processed for {checkpoint}", "blue")) - checkpoint = item['checkpoint'] + array_dividend = max(round(len(histories) / 20), 1) + match_found = False + for index, item in enumerate(histories[::array_dividend]): + if (datetime.date.today() - datetime.timedelta(days=days) <= datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()): + match_found = True break - else: - if (datetime.date.today() - datetime.timedelta(days=days) > datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()): - pass - else: json_output['response']['exechistories'].append(item) - seen = {} - if os.path.exists(file_path): - with open(file_path, 'r') as file: - existing_data = json.load(file) - combined = existing_data['response']['exechistories'] + json_output['response']['exechistories'] - else: - combined = json_output['response']['exechistories'] - for item in combined: - key = (item.get('sha256'), item.get('filename'), item.get('hostname')) - seen[key] = item - deduplicated = list(seen.values()) - with open(file_path, 'w') as file: - json.dump({'error': 'Success', 'response': {'exechistories': deduplicated}}, file) - json_output['response']['exechistories'].clear() + if match_found == True: + for index, item in enumerate(tqdm.tqdm(histories, desc=f"Processing events for {checkpoint}", unit="events", colour="blue", initial=1)): + if index == len(histories) - 1: + checkpoint = item['checkpoint'] + break + else: + if (datetime.date.today() - datetime.timedelta(days=days) > datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()): + pass + else: json_output['response']['exechistories'].append(item) + seen = {} + if os.path.exists(file_path): + with open(file_path, 'r') as file: + existing_data = json.load(file) + combined = existing_data['response']['exechistories'] + json_output['response']['exechistories'] + else: + combined = json_output['response']['exechistories'] + for item in combined: + key = (item.get('sha256'), item.get('filename'), item.get('hostname')) + seen[key] = item + deduplicated = list(seen.values()) + with open(file_path, 'w') as file: + json.dump({'error': 'Success', 'response': {'exechistories': deduplicated}}, file) + json_output['response']['exechistories'].clear() + date_diff = datetime.date.today() - datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date() + percentage_diff = ((days - date_diff.days) / 60) * 100 + pbar.n = round(percentage_diff) + pbar.set_description_str(f"Total Percentage Complete: ({percentage_diff:.1f}%)") + pbar.refresh with open(file_path, 'r') as file: final_output = json.load(file) os.remove(file_path)