Merge pull request 'IOTesting' (#19) from IOTesting into master

Reviewed-on: brotoskyj/AirlockTools#19
This commit was merged in pull request #19.
This commit is contained in:
brotoskyj
2025-09-05 11:02:43 -04:00
7 changed files with 599 additions and 561 deletions
+36 -417
View File
@@ -14,8 +14,6 @@
# along with this program. If not, see <https://www.gnu.org/licenses/>.
import dotenv
import gc
import json
import os
import pandas as pd
import urllib3
@@ -26,19 +24,20 @@ import utils.pathfunctions
import utils.policyfunctions
import utils.pretty as ct
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
dotenv.load_dotenv()
#Constants
url = os.getenv('url')
badpublisherlist = ["Brave Software, Inc.", "Zoom Video Communications, Inc."]
badpathparts = ["users", "inet\\wwwroot", "windows\\temp", "windows\\temp", "windows\\task", "windows\\system32", "startup", "windows\\fonts", "Recycle.Bin", "AppData", "programdata"]
path_exclusion_constant = 3
bad_publisher_list = ["Brave","Zoom", "GlavSoft", "VNC"]
pups = ["logmein", "invalid"]
badpathparts = ["users", "wwwroot", "windows\\temp", "windows\\task", "windows\\system32", "startup", "windows\\fonts", "Recycle.Bin", "AppData", "programdata"]
path_exclusion_constant = 4
min_files_for_path = 4
threat_tolerance_constant = 4
def apivalidation():
match os.getenv('APIKEY'):
case '':
@@ -137,10 +136,12 @@ def menu_prepare_to_enforce():
first_policy = " "
second_policy = " "
allowlist_parent_name = " "
allowlist_child_name = " "
destination_name = " "
df_aggregated_combo = pd.DataFrame()
destination_id = " "
allowlist_parent_name = " "
allowlist_parent_id = " "
allowlist_child_name = " "
allowlist_child_id = " "
#If the directorys where we're going to store our output dont exist, make them.
if not os.path.exists("parquet"): os.makedirs("parquet")
@@ -149,119 +150,8 @@ def menu_prepare_to_enforce():
if not os.path.exists("preflight"): os.makedirs("preflight")
while True:
print(ct.colorText("\n --------------------------------------------------------------------", "cyan"))
print(ct.colorText(" -------------------- Prepare to Enforce Policy ---------------------", "cyan"))
print(ct.colorText(" --------------------------------------------------------------------", "cyan"))
print(ct.colorText("\nSequentually follow these steps to prepare a policy for enforcement:", "white"))
print(ct.colorText("\n1. Choose which originating policy or policies to move to enforcement", "cyan"))
if first_policy == " " and second_policy == " ":
print(ct.colorText(f" [✗] No policies have been chosen","red"))
elif first_policy != " " and second_policy is first_policy:
print(ct.colorText(f" [✓] {first_policy} has been selected,", "green"))
elif first_policy != " " and second_policy != " ":
print(ct.colorText(f" [✓] {first_policy} has been selected as Policy 1","green"))
print(ct.colorText(f" [✓] {second_policy} has been selected as Policy 2","green"))
print(ct.colorText("2. Pull and stage event history, combine the histories, add hash info, then categorize the hashes", "cyan"))
if os.path.exists(f"parquet\\execution_history_{first_policy}.parquet"):
print(ct.colorText(f" [✓] Execution history has been compiled for {first_policy}","green"))
elif not os.path.exists(f"parquet\\execution_history_{first_policy}.parquet"):
print(ct.colorText(f" [✗] Execution history has not been compiled for {first_policy}","red"))
elif second_policy is not first_policy and os.path.exists(f"parquet\\execution_history_{second_policy}.parquet"):
print(ct.colorText(f" [✓] Execution history has been compiled for {second_policy}","green"))
elif second_policy is not first_policy and not os.path.exists(f"parquet\\execution_history_{second_policy}.parquet"):
print(ct.colorText(f" [✗] Execution history has not been compiled for {second_policy}","red"))
if os.path.exists(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"):
print(ct.colorText(f" [✓] Hash Info has been added to the combined execution history", "green"))
else:
print(ct.colorText(f" [✗] Hash Info has not been added to the combined execution history", "red"))
if os.path.exists(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") and os.path.exists(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") and os.path.exists(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet"):
print(ct.colorText(f" [✓] Hashes have been cateogrized", "green"))
else:
print(ct.colorText(f" [✗] Hashes have not been cateogrized", "red"))
if os.path.exists(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet"):
print(ct.colorText(f" [✓] Execution history has been_combined_for {first_policy} and_{second_policy}", "green"))
else:
print(ct.colorText(f" [✗] Execution history has not been_combined_for {first_policy} and_{second_policy}", "red"))
print(ct.colorText(f"3. Manually review the files:","cyan"))
print(ct.colorText(" 'needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv' and 'needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv'", "cyan"))
print(ct.colorText(" Remove the rows containing hashes you do not approve of, and those you would not approve of without metarules.", "cyan"))
print(ct.colorText(" If metarules need to be created, please make note of them, and remove the row from the csv.", "cyan"))
print(ct.colorText(" When complete, save both csv files to the directory 'approved' and choose this option.","cyan"))
print(ct.colorText(" This will combine these approved hashes with the automatically approved hashes and generate a list of paths to be reviewed", "cyan"))
if os.path.exists(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv") and os.path.exists(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv"):
print(ct.colorText(" [✓] Reviewed hashes have been loaded","green"))
else:
print(ct.colorText(" [✗] Reviewed hashes have not been loaded","red"))
if os.path.exists(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet"):
print(ct.colorText(" [✓] The combined approved hashes list has been generated","green"))
else:
print(ct.colorText(" [✗] The combined approved hashes list has not been generated","red"))
if os.path.exists(f"parquet\\approved_hashes_with_paths_{first_policy}_{second_policy}.parquet"):
print(ct.colorText(" [✓] Longest common filepaths have been generated and appended to hash info","green"))
else:
print(ct.colorText(" [✗] Longest common filepaths have not been generated","red"))
if os.path.exists(f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.csv"):
print(ct.colorText(" [✓] Path review list created","green"))
else:
print(ct.colorText(" [✗] Path review list has not been created","red"))
print(ct.colorText(f"4. Manually review the file 'needs_approved\\paths_needing_review_{first_policy}_{second_policy}.csv'", "cyan"))
print(ct.colorText(" Remove the rows containing path exclusions you do not approve of" , "cyan"))
print(ct.colorText(" When complete, save the csv file to the directory 'approved'", "cyan"))
print(ct.colorText(" Preflight Lists will be generated", "cyan"))
if os.path.exists(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv"):
print(ct.colorText(" [✓] Reviewed path list detected","green"))
else:
print(ct.colorText(" [✗] Path review list has not been detected","red"))
if os.path.exists(f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html"):
print(ct.colorText(" [✓] Preflight Path Exclusion List has been generated","green"))
else:
print(ct.colorText(" [✗] Preflight Path Exclusion List has not been generated","red"))
if os.path.exists(f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html"):
print(ct.colorText(" [✓] Preflight hash approval list has been generated","green"))
else:
print(ct.colorText(" [✗] Preflight hash approval list has not been generated","red"))
print(ct.colorText(f"5. Choose the destination policy and parent and child allow list", "cyan"))
if allowlist_child_name == " " and allowlist_parent_name== " ":
print(ct.colorText(f" [✗] No allowlists have been chosen","red"))
elif allowlist_parent_name != " " and allowlist_child_name != " " and allowlist_parent_name is allowlist_child_name:
print(ct.colorText(f" [✓] [✗] Only {allowlist_parent_name} has been selected this is unusual, but potentially valid case, double check before proceeding,", "yellow"))
elif allowlist_parent_name != " " and allowlist_child_name != " " and allowlist_parent_name is not allowlist_child_name:
print(ct.colorText(f" [✓] {allowlist_parent_name} has been selected as Parent Policy","green"))
print(ct.colorText(f" [✓] {allowlist_child_name} has been selected as Child Policy","green"))
if destination_name == " ":
print(ct.colorText(f" [✗] No destination policy has been chosen","red"))
else:
print(ct.colorText(f" [✓] destination policy is {destination_name}","green"))
print(ct.colorText(f"6. Liftoff ------------------------------------------------------", "cyan"))
print(ct.colorText(f" Apply path exclusions according to allowed and approved paths", "cyan"))
print(ct.colorText(f" Apply signed or attested hashes to Parent Allow List", "cyan"))
print(ct.colorText(f" Apply approved, but unsigned hashes to the Child Allow List", "cyan"))
print(ct.colorText("Q. Quit", "cyan"))
ct.printEnforceChecklist(first_policy, second_policy, allowlist_child_name, allowlist_parent_name, destination_name)
choice = input(ct.colorText("\nEnter your choice: ", "white"))
@@ -285,296 +175,42 @@ def menu_prepare_to_enforce():
elif choice == "2":
if not os.path.exists(f"parquet\\execution_history_{first_policy}.parquet"):
print(choice)
print(first_policy)
exe1 = utils.allowlist.pullPolicyExechistories(url, first_policy, 60, True)
data = json.loads(exe1)
executionhist_policy1 = pd.DataFrame(data["response"]["exechistories"])
if not executionhist_policy1.empty:
executionhist_policy1 = executionhist_policy1[['sha256', 'publisher', 'filename', 'hostname', 'username', 'pprocess', 'gprocess', 'commandline']]
executionhist_policy1 = executionhist_policy1.drop_duplicates(subset=['sha256', 'filename', 'hostname'])
executionhist_policy1 = executionhist_policy1.sort_values(by=['sha256', 'filename'])
executionhist_policy1.to_parquet(f"parquet\\execution_history_{first_policy}.parquet", index=False)
print(ct.colorText(f"Staging of Execution history for policy: {first_policy} is complete", "green"))
del data
del exe1
del executionhist_policy1
gc.collect()
utils.policyfunctions.getPolicyInfo(url, first_policy, 60)
if not os.path.exists(f"parquet\\execution_history_{second_policy}.parquet"):
exe2 = utils.allowlist.pullPolicyExechistories(url,second_policy, 60, True)
data2 = json.loads(exe2)
executionhist_policy2 = pd.DataFrame(data2["response"]["exechistories"])
if not executionhist_policy2.empty:
executionhist_policy2 = executionhist_policy2[['sha256', 'publisher', 'filename', 'hostname', 'username', 'pprocess', 'gprocess', 'commandline']]
executionhist_policy2 = executionhist_policy2.drop_duplicates(subset=['sha256', 'filename', 'hostname'])
executionhist_policy2 = executionhist_policy2.sort_values(by=['sha256', 'filename'])
executionhist_policy2.to_parquet(f"parquet\\execution_history_{second_policy}.parquet", index=False)
print(ct.colorText(f"Staging of Execution history for policy: {second_policy} is complete", "green"))
del executionhist_policy2
del data2
del exe2
gc.collect()
utils.policyfunctions.getPolicyInfo(url, second_policy, 60)
if not os.path.exists(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"):
combined_hashes = pd.DataFrame(columns=['sha256', 'publisher'])
hashes = []
try:
hash1 = pd.read_parquet(f"parquet\\execution_history_{first_policy}.parquet", columns=['sha256', 'publisher'])
utils.pathfunctions.inspect_parquet(f"parquet\\execution_history_{first_policy}.parquet")
if not hash1.empty:
hashes.append(hash1)
else:
print("⚠️ First dataframe is empty.")
except Exception as e:
print(f"❌ Error reading first Parquet file: {e}")
try:
hash2 = pd.read_parquet(f"parquet\\execution_history_{second_policy}.parquet", columns=['sha256', 'publisher'])
utils.pathfunctions.inspect_parquet(f"parquet\\execution_history_{second_policy}.parquet")
if not hash2.empty:
hashes.append(hash2)
else:
print("⚠️ Second dataframe is empty.")
except Exception as e:
print(f"❌ Error reading second Parquet file: {e}")
if hashes:
combined_hashes = pd.concat(hashes, ignore_index=True)
print(f"✅ Combined {len(combined_hashes)} hashes.")
else:
print("⚠️ No valid dataframes to combine.")
combined_hashes = combined_hashes.drop_duplicates(subset=['sha256'])
augmented_combo = utils.hashfunctions.augmentAggregatedHashes(url, combined_hashes)
numeric_reputation_cols = [
'reputation_scannermatch',
'reputation_scannercount',
'reputation_threatlevel'
]
for col in numeric_reputation_cols:
if col in augmented_combo.columns:
augmented_combo[col] = pd.to_numeric(augmented_combo[col].replace('N/A', pd.NA), errors='coerce')
augmented_combo = augmented_combo.rename(columns={'publisher_x': 'publisher'})
augmented_combo = augmented_combo[['sha256', 'publisher', 'description', 'productname', 'productversion',
'reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount',
'reputation_status', 'reputation_threatlevel', 'reputation_threatname',
'reputation_timestamp']]
augmented_combo = augmented_combo.sort_values(by=['publisher', 'description', 'productname'])
augmented_combo.to_parquet(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet", index=False)
del combined_hashes
del augmented_combo
gc.collect()
print(ct.colorText("Hash reputation info added to dataframe", "green"))
utils.hashfunctions.combineHashes(url, first_policy, second_policy)
if not os.path.exists(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet"):
# Categorize the hashes
categorized = utils.hashfunctions.categorizeHashes(
utils.hashfunctions.categorizeHashes(
first_policy,
second_policy,
pd.read_parquet(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"),
threat_tolerance_constant,
badpublisherlist
bad_publisher_list,
pups
)
categorized[0].to_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet", index=False)
categorized[1].to_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", index=False)
categorized[2].to_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", index=False)
del categorized
gc.collect()
if not os.path.exists(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet"):
# Condense execution history
try:
exe1 = pd.read_parquet(f"parquet\\execution_history_{first_policy}.parquet")
utils.pathfunctions.inspect_parquet(f"parquet\\execution_history_{first_policy}.parquet")
if not exe1.empty:
condensed_exe1 = exe1.groupby('sha256').agg(lambda x: list(set(x))).reset_index()
else:
print("⚠️ First dataframe is empty.")
except Exception as e:
print(f"❌ Error reading first Parquet file: {e}")
utils.hashfunctions.condenseExecutions(first_policy,second_policy)
try:
exe2 = pd.read_parquet(f"parquet\\execution_history_{second_policy}.parquet")
utils.pathfunctions.inspect_parquet(f"parquet\\execution_history_{second_policy}.parquet")
if not exe2.empty:
condensed_exe2 = exe2.groupby('sha256').agg(lambda x: list(set(x))).reset_index()
else:
print("⚠️ Second dataframe is empty.")
except Exception as e:
print(f"❌ Error reading second Parquet file: {e}")
if not exe1.empty and not exe2.empty:
condensed_combo = pd.concat([condensed_exe1, condensed_exe2], ignore_index=True)
print(f"✅ Combined {len(condensed_combo)} hashes.")
elif exe1.empty:
condensed_combo = condensed_exe2
elif exe2.empty:
condensed_combo = condensed_exe1
else:
print("⚠️ No valid dataframes to combine.")
condensed_combo.to_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet", index=False)
del condensed_combo
gc.collect()
if not os.path.exists(f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv"):
condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet")
needsapproval= pd.read_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet")
#Pull hash info for the entries in the needs approval table
needsapproval = pd.merge(condensed_combo, needsapproval, on='sha256', how='inner')
#Deduplicate lists in the columns
for col in needsapproval.columns:
if needsapproval[col].apply(lambda x: isinstance(x, list)).all():
needsapproval[col] = needsapproval[col].apply(deduplicate_list)
#Rename Publisher, Keep and reorder columns we want
needsapproval = needsapproval.rename(columns={'publisher_x': 'publisher'})
needsapproval = needsapproval[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']]
needsapproval.to_csv(f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv",index=False)
needsapproval.to_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet",index=False)
ct.style_dataframe_dark(needsapproval, f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.html")
del needsapproval
del condensed_combo
gc.collect()
if not os.path.exists(f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv"):
condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet")
needsapproval= pd.read_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet")
#Pull hash info for the entries in the needs approval table
needsapproval = pd.merge(condensed_combo, needsapproval, on='sha256', how='inner')
#Deduplicate lists in the columns
for col in needsapproval.columns:
if needsapproval[col].apply(lambda x: isinstance(x, list)).all():
needsapproval[col] = needsapproval[col].apply(deduplicate_list)
#Rename Publisher, Keep and reorder columns we want
needsapproval = needsapproval.rename(columns={'publisher_x': 'publisher'})
needsapproval = needsapproval[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']]
needsapproval.to_csv(f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv",index=False)
needsapproval.to_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", index=False)
ct.style_dataframe_dark(needsapproval, f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.html")
del needsapproval
del condensed_combo
gc.collect()
if not os.path.exists(f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.html"):
condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet")
needsapproval= pd.read_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet")
#Pull hash info for the entries in the needs approval table
needsapproval = pd.merge(condensed_combo, needsapproval, on='sha256', how='inner')
#Deduplicate lists in the columns
for col in needsapproval.columns:
if needsapproval[col].apply(lambda x: isinstance(x, list)).all():
needsapproval[col] = needsapproval[col].apply(deduplicate_list)
#Rename Publisher, Keep and reorder columns we want
needsapproval = needsapproval.rename(columns={'publisher_x': 'publisher'})
needsapproval = needsapproval[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']]
needsapproval.to_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", index=False)
ct.style_dataframe_dark(needsapproval, f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.html")
del needsapproval
del condensed_combo
gc.collect()
if os.path.exists(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") & os.path.exists(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") & os.path.exists(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet"):
utils.hashfunctions.divideSortedHashExecutions(first_policy,second_policy,pups)
elif choice == "3":
if os.path.exists(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv") and os.path.exists(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv"):
if not os.path.exists(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet"):
df1 = tryToReadCSV(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv")
df2 = tryToReadCSV(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv")
all_approved_hashes = pd.concat([df1 , df2], ignore_index=True).sort_values(by=['sha256','filename'])
print(ct.colorText(f"Approved hash lists have been combined","green"))
all_approved_hashes.to_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet", index=False)
del all_approved_hashes
gc.collect()
if not os.path.exists(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet"):
all_approved_hashes = pd.read_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet")
print(ct.colorText(f"Beginning calculating longest common filepaths for path exceptions","green"))
grouped_df_view, df_with_groups_appended = utils.pathfunctions.export_groups_for_review(all_approved_hashes,"filename","longestcfp",min_files_for_path,path_exclusion_constant)
df_with_groups_appended.to_parquet(f"parquet\\approved_hashes_with_paths_{first_policy}_{second_policy}.parquet", index=False)
forbidden = utils.pathfunctions.regulator(badpathparts, True)
forbidden_lcfp = grouped_df_view["longestcfp"].str.contains(forbidden, na=False)
grouped_df_view = grouped_df_view[~forbidden_lcfp]
print(ct.colorText(f"Removing forbidden filepaths for path exceptions","green"))
grouped_df_view.to_parquet(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet", index=False)
grouped_df_view.to_csv(f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.csv", index=False)
ct.style_dataframe_dark(grouped_df_view, f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.html")
del grouped_df_view
del df_with_groups_appended
gc.collect()
utils.pathfunctions.generatePathReview(first_policy, second_policy, badpathparts, min_files_for_path)
else:
print(ct.colorText(f"Please manually approve hashes prior to this step","red"))
elif choice == "4":
if os.path.exists(f"parquet\\approved_hashes_with_paths_{first_policy}_{second_policy}.parquet") and os.path.exists(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv"):
if os.path.exists(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv"):
if not os.path.exists(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet"):
df1 = pd.read_parquet(f"parquet\\approved_hashes_with_paths_{first_policy}_{second_policy}.parquet")
allowbyhash = utils.pathfunctions.mask_from_csv(df1, f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv","longestcfp")
pathexclusions = tryToReadCSV(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv")
pathexclusions.to_parquet(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet", index=False)
allowbyhash.to_parquet(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet", index=False)
easyview = allowbyhash.groupby('sha256').agg(list).reset_index()
ct.style_dataframe_dark(easyview, f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html")
ct.style_dataframe_dark(pathexclusions, f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html")
del df1
del allowbyhash
del pathexclusions
del easyview
gc.collect()
utils.hashfunctions.generatePreflights(first_policy, second_policy)
elif choice == "5":
@@ -598,34 +234,17 @@ def menu_prepare_to_enforce():
elif choice == "6":
if os.path.exists(f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html") and os.path.exists(f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html") and allowlist_parent_name != " " and allowlist_child_name != " " and destination_name != " ":
pathexclusions = pd.read_parquet(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet")
allowbyhash = pd.read_parquet(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet")
ct.areYouSure()
confirmation = input(ct.colorText("Type 'I AGREE' to continue: ","white"))
if confirmation.strip().upper() == "I AGREE":
print(ct.colorText("Proceeding with the code...", "yellow"))
print(ct.colorText(f"Adding path exclusions to {destination_name}", "yellow"))
pathexcludelist = pathexclusions['longestcfp'].unique().tolist()
utils.policyfunctions.addPath(destination_id,pathexcludelist)
print(ct.colorText(f"Adding hashes to {allowlist_parent_name}", "yellow"))
allowlist_parenthashlist = allowbyhash[allowbyhash['reputation_status'] == 'KNOWN']['sha256'].unique().tolist()
utils.policyfunctions.addHash(allowlist_parent_id,allowlist_parenthashlist)
print(ct.colorText(f"Adding hashes to {allowlist_child_name}", "yellow"))
allowlist_childhashlist = allowbyhash[allowbyhash['reputation_status'] == 'UNKNOWN']['sha256'].unique().tolist()
utils.policyfunctions.addHash(allowlist_child_id, allowlist_childhashlist)
ct.locked()
exit()
else:
print(ct.colorText("Operation aborted. You MUST EXPLICITLY AGREE to proceed.", "red"))
break
utils.policyfunctions.sendToPolicy(
url,
first_policy,
second_policy,
destination_name,
destination_id,
allowlist_parent_name,
allowlist_parent_id,
allowlist_child_name,
allowlist_child_id
)
elif choice == "Q":
break
+27 -2
View File
@@ -1,4 +1,29 @@
pandas==2.3.2
bson==0.5.10
certifi==2025.8.3
charset-normalizer==3.4.3
colorama==0.4.6
cramjam==2.11.0
docopt==0.6.2
dotenv==0.9.9
fastparquet==2024.11.0
fsspec==2025.9.0
idna==3.10
ijson==3.4.0
lxml==6.0.0
markdown-it-py==4.0.0
mdurl==0.1.2
numpy==2.3.2
packaging==25.0
pandas==2.3.1
pretty-tables==3.1.0
pyarrow==21.0.0
Pygments==2.19.2
python-dateutil==2.9.0.post0
python-dotenv==1.1.1
Requests==2.32.5
pytz==2025.2
requests==2.32.4
six==1.17.0
tqdm==4.67.1
tzdata==2025.2
urllib3==2.5.0
yarg==0.1.10
+15 -12
View File
@@ -21,6 +21,8 @@ import ijson
import os
from bson import ObjectId
import datetime
import tqdm
import sys
def pullPolicyExechistories(url, policiesnames, days, outputjson: bool):
file_path = 'chunkinator.json'
@@ -33,32 +35,27 @@ def pullPolicyExechistories(url, policiesnames, days, outputjson: bool):
headers = {"X-APIKey": os.getenv('APIKEY')}
checkpoint = str(skipback(days))
json_output = {'error': 'Success', 'response': {'exechistories': []}}
with tqdm.tqdm(file=sys.stdout, leave=True, total=10000, desc=f"Checkpoint Progess: {checkpoint}", colour="blue", initial=1) as filebar:
with tqdm.tqdm(file=sys.stdout, leave=True, total=100, desc=f"Total of {policiesnames} Complete: ") as pbar:
while True:
json_response_data = checkpoint_stomper(checkpoint, url, policiesnames, headers)
histories = json_response_data['response']['exechistories']
filebar.total=len(histories)
if not histories:
break
array_dividend = max(round(len(histories) / 20), 1)
match_found = False
for index, item in enumerate(histories[::array_dividend]):
if (datetime.date.today() - datetime.timedelta(days=days) <= datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()):
match_found = True
break
checkpoints_processed = round(len(histories) / array_dividend)
if ( index + 1 ) < checkpoints_processed:
print(ct.colorText(f"{index + 1}/{checkpoints_processed} checkpoint(s) from this execution have been processed with date match. Last Checkpoint: {checkpoint}", "blue"))
else:
print(ct.colorText(f"{index + 1}/{checkpoints_processed} checkpoint(s) Processed. Last Checkpoint: {checkpoint}", "blue"))
if match_found == True:
for index, item in enumerate(histories):
if index == len(histories) - 1:
print(ct.colorText(f"All Events Processed for {checkpoint}", "blue"))
checkpoint = item['checkpoint']
filebar.desc = f"Checkpoint Progress: {checkpoint}"
break
else:
if (datetime.date.today() - datetime.timedelta(days=days) > datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()):
pass
else: json_output['response']['exechistories'].append(item)
filebar.update(1)
filebar.refresh()
seen = {}
if os.path.exists(file_path):
with open(file_path, 'r') as file:
@@ -73,6 +70,12 @@ def pullPolicyExechistories(url, policiesnames, days, outputjson: bool):
with open(file_path, 'w') as file:
json.dump({'error': 'Success', 'response': {'exechistories': deduplicated}}, file)
json_output['response']['exechistories'].clear()
date_diff = datetime.date.today() - datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()
percentage_diff = (((days + 10) - date_diff.days) / (days + 10)) * 100
pbar.n = round(percentage_diff)
pbar.set_description_str(f"Total of {policiesnames} Complete: ")
pbar.refresh()
filebar.n = 1
with open(file_path, 'r') as file:
final_output = json.load(file)
os.remove(file_path)
@@ -144,7 +147,7 @@ def skipback(days):
Generate a MongoDB ObjectId for a given number of days ago from today.
Adds 1 extra day to the input to look further back.
"""
adjusted_days = days + 1
adjusted_days = days + 10
date_days_ago = datetime.datetime.now(datetime.UTC) - datetime.timedelta(days=adjusted_days)
timestamp = int(date_days_ago.timestamp())
hex_timestamp = format(timestamp, '08x')
+187 -13
View File
@@ -12,12 +12,15 @@
#
# You should have received a copy of the GNU Affero General Public License
# along with this program. If not, see <https://www.gnu.org/licenses/>.
import gc
import json
import os
import pandas as pd
import requests
import os
import json
import utils.pathfunctions as pathf
import utils.hashfunctions as hashf
import utils.pretty as ct
from AirlockTools import tryToReadCSV
def aggregateHashes(executions_json) -> pd.DataFrame:
"""
@@ -101,11 +104,9 @@ def augmentAggregatedHashes(url, agg_df: pd.DataFrame) -> pd.DataFrame:
return aug_df
def categorizeHashes(aug_df: pd.DataFrame, threat_tolerance: int, untrusted_publishers: list):
if untrusted_publishers is None:
untrusted_publishers = []
df = aug_df.copy()
def categorizeHashes(first_policy, second_policy, df: pd.DataFrame, threat_tolerance: int, untrusted_publishers, pups: list):
if untrusted_publishers is None: untrusted_publishers = []
if pups is None: pups = []
def reputationtool(row):
val = row["reputation_scannermatch"]
@@ -126,14 +127,16 @@ def categorizeHashes(aug_df: pd.DataFrame, threat_tolerance: int, untrusted_publ
mask_approved = (
(
(df["publisher"] != "Not Signed") &
~df["publisher"].isin(untrusted_publishers) &
~df["reputation_status"].isna()
~df["publisher"].str.contains(pathf.regulator(untrusted_publishers), case=False, na=False) &
~df["reputation_status"].isna() &
~df["description"].str.contains(pathf.regulator(pups), case=False, na=False)
) |
(
(df["publisher"] == "Not Signed") &
~df["reputation_flag"] &
~df["publisher"].isin(untrusted_publishers) &
~df["reputation_status"].isna()
~df["publisher"].str.contains(pathf.regulator(untrusted_publishers), case=False, na=False) &
~df["reputation_status"].isna() &
~df["description"].str.contains(pathf.regulator(pups), case=False, na=False)
)
)
@@ -141,7 +144,14 @@ def categorizeHashes(aug_df: pd.DataFrame, threat_tolerance: int, untrusted_publ
approved_df = df[mask_approved]
unapproved_df = df[~(mask_needsreview | mask_approved)]
return needsreview_df, approved_df, unapproved_df
needsreview_df.to_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet", index=False)
approved_df.to_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", index=False)
unapproved_df.to_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", index=False)
del needsreview_df
del approved_df
del unapproved_df
gc.collect()
def explode_and_deduplicate(df):
df['sha256'] = df['sha256'].str.split(',')
@@ -199,3 +209,167 @@ def destinationHashes(
# Concatenate results
df_hashdestination = pd.concat([df_paths, df_hashes], ignore_index=True)
return df_hashdestination
def combineHashAndHist(path, first_policy, second_policy):
condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet")
df = pd.read_parquet(path)
#Pull hash info for the entries in the needs approval table
df = pd.merge(condensed_combo, df, on='sha256', how='inner')
#Rename Publisher, Keep and reorder columns we want
df = df.rename(columns={'publisher_x': 'publisher'})
df = df[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']]
df = df.sort_values(by='filename')
df.to_parquet(path, index=False)
del df
del condensed_combo
gc.collect()
def combineHashes(url, first_policy, second_policy):
combined_hashes = pd.DataFrame(columns=['sha256', 'publisher'])
hashes = []
try:
hash1 = pd.read_parquet(f"parquet\\execution_history_{first_policy}.parquet", columns=['sha256', 'publisher'])
pathf.inspect_parquet(f"parquet\\execution_history_{first_policy}.parquet")
if not hash1.empty:
hashes.append(hash1)
else:
print("⚠️ First dataframe is empty.")
except Exception as e:
print(f"❌ Error reading first Parquet file: {e}")
try:
hash2 = pd.read_parquet(f"parquet\\execution_history_{second_policy}.parquet", columns=['sha256', 'publisher'])
pathf.inspect_parquet(f"parquet\\execution_history_{second_policy}.parquet")
if not hash2.empty:
hashes.append(hash2)
else:
print("⚠️ Second dataframe is empty.")
except Exception as e:
print(f"❌ Error reading second Parquet file: {e}")
if hashes:
combined_hashes = pd.concat(hashes, ignore_index=True)
print(f"✅ Combined {len(combined_hashes)} hashes.")
else:
print("⚠️ No valid dataframes to combine.")
combined_hashes = combined_hashes.drop_duplicates(subset=['sha256'])
augmented_combo = hashf.augmentAggregatedHashes(url, combined_hashes)
numeric_reputation_cols = [
'reputation_scannermatch',
'reputation_scannercount',
'reputation_threatlevel'
]
for col in numeric_reputation_cols:
if col in augmented_combo.columns:
augmented_combo[col] = pd.to_numeric(augmented_combo[col].replace('N/A', pd.NA), errors='coerce')
augmented_combo = augmented_combo.rename(columns={'publisher_x': 'publisher'})
augmented_combo = augmented_combo[['sha256', 'publisher', 'description', 'productname', 'productversion',
'reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount',
'reputation_status', 'reputation_threatlevel', 'reputation_threatname',
'reputation_timestamp']]
augmented_combo = augmented_combo.sort_values(by=['publisher', 'description', 'productname'])
augmented_combo.to_parquet(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet", index=False)
del combined_hashes
del augmented_combo
gc.collect()
print(ct.colorText("Hash reputation info added to dataframe", "green"))
def condenseExecutions(first_policy,second_policy):
exe1 = pd.DataFrame()
exe2 = pd.DataFrame()
condensed_combo = pd.DataFrame()
try:
exe1 = pd.read_parquet(f"parquet\\execution_history_{first_policy}.parquet")
pathf.inspect_parquet(f"parquet\\execution_history_{first_policy}.parquet")
if not exe1.empty:
print()
else:
print("⚠️ First dataframe is empty.")
except Exception as e:
print(f"❌ Error reading first Parquet file: {e}")
try:
exe2 = pd.read_parquet(f"parquet\\execution_history_{second_policy}.parquet")
pathf.inspect_parquet(f"parquet\\execution_history_{second_policy}.parquet")
if not exe2.empty:
print()
else:
print("⚠️ Second dataframe is empty.")
except Exception as e:
print(f"❌ Error reading second Parquet file: {e}")
if not exe1.empty and not exe2.empty:
condensed_combo = pd.concat([exe1, exe2], ignore_index=True)
print(f"✅ Combined {len(condensed_combo)} hashes.")
elif exe1.empty:
condensed_combo = exe2
elif exe2.empty:
condensed_combo = exe1
else:
print("⚠️ No valid dataframes to combine.")
condensed_combo.to_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet", index=False)
del condensed_combo
gc.collect()
def divideSortedHashExecutions(first_policy,second_policy, pups):
combineHashAndHist(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet", first_policy, second_policy)
combineHashAndHist(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", first_policy, second_policy)
combineHashAndHist(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", first_policy, second_policy)
unknown = pd.read_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet")
good = pd.read_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet")
bad = pd.read_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet")
# Build regex pattern once
pattern = pathf.regulator(pups)
# Move matching rows from unknown and good to bad
bad = pd.concat([
bad,
unknown[unknown["filename"].str.contains(pattern, na=False)],
good[good["filename"].str.contains(pattern, na=False)]
], ignore_index=True)
# Remove matching rows from unknown and good
unknown = unknown[~unknown["filename"].str.contains(pattern, na=False)]
good = good[~good["filename"].str.contains(pattern, na=False)]
unknown.to_csv(f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv",index=False)
good.to_csv(f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv",index=False)
bad.to_csv(f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.csv",index=False)
ct.style_dataframe_dark(unknown, f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.html")
ct.style_dataframe_dark(good, f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.html")
ct.style_dataframe_dark(bad, f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.html")
def generatePreflights(first_policy, second_policy):
allhashes = pd.read_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet")
pathexclusions = tryToReadCSV(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv")
pathexclusions.to_parquet(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet", index=False)
allowbyhash = allhashes[~allhashes['sha256'].isin(pathexclusions['sha256'])]
allowbyhash.to_parquet(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet", index=False)
allowbyhash.sort_values(by=["filename"])
ct.style_dataframe_dark(allowbyhash, f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html")
ct.style_dataframe_dark(pathexclusions, f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html")
del allowbyhash
del pathexclusions
gc.collect()
+96 -64
View File
@@ -12,76 +12,61 @@
#
# You should have received a copy of the GNU Affero General Public License
# along with this program. If not, see <https://www.gnu.org/licenses/>.
import pandas as pd
import os
from itertools import chain
import ast
import gc
import os
import pandas as pd
import re
import utils.pathfunctions as pathf
import utils.pretty as ct
from AirlockTools import tryToReadCSV
def split_path(path):
parts = []
while True:
head, tail = os.path.split(path)
if tail:
parts.insert(0, tail)
path = head
else:
if head:
parts.insert(0, head)
break
def split_filepaths_grouped(df, col="filename", group_parts=4, min_parts=4):
def clean_split(path):
parts = os.path.normpath(path).split(os.sep)
# Remove leading empty strings caused by UNC paths
parts = [p for p in parts if p]
return parts
def local_common_pass(paths, min_parts=3):
results = {}
paths_sorted = sorted(paths)
for i, path in enumerate(paths_sorted):
candidates = []
df = df.copy()
split_paths = df[col].apply(clean_split)
if i > 0:
try:
candidates.append(os.path.commonpath([path, paths_sorted[i-1]]))
except ValueError:
# different drives, skip
pass
if i < len(paths_sorted) - 1:
try:
candidates.append(os.path.commonpath([path, paths_sorted[i+1]]))
except ValueError:
# different drives, skip
pass
# Filter out paths with fewer than `min_parts` components
df = df[split_paths.apply(lambda parts: len(parts) >= min_parts)].copy()
split_paths = split_paths[df.index] # Update split_paths to match filtered df
best = path
best_len = 0
for c in candidates:
parts = split_path(c)
if len(parts) >= min_parts and len(parts) > best_len:
best = c
best_len = len(parts)
results[path] = best
return results
df["group_key"] = split_paths.apply(lambda parts: os.sep.join(parts[:group_parts]))
grouped = df.groupby("group_key")
new_rows = []
def add_longest_common_two_local(df, col="filename_x", new_col="longestcfp", min_parts=3):
dirs_series = df[col].astype(str).apply(os.path.dirname)
first_pass = local_common_pass(dirs_series.tolist(), min_parts)
second_pass = local_common_pass(list(first_pass.values()), min_parts)
df[new_col] = dirs_series.map(lambda d: second_pass[first_pass[d]])
return df
for _, group_df in grouped:
paths = group_df[col].tolist()
split_parts = [clean_split(p) for p in paths]
def export_groups_for_review(df, col, group_col, min_number_in_group, path_length_constant):
"""
Compute longest common paths, group filepaths, write CSV for review.
"""
df = df.drop_duplicates(subset=[col], keep='first')
df = add_longest_common_two_local(df, col=col, new_col=group_col)
grouped = df.groupby(group_col)[col].apply(list).reset_index()
grouped = grouped.sort_values(by=col)
print("Before filtering:", len(grouped))
grouped = grouped[grouped[col].apply(lambda x: len(x) >= min_number_in_group)]
filtered = grouped[grouped[group_col].apply(lambda x: len(os.path.normpath(x).split(os.sep)) >= path_length_constant)]
print("After filtering:", len(grouped))
def longest_common_prefix(paths):
if not paths:
return []
prefix = paths[0]
for path in paths[1:]:
prefix = [a for a, b in zip(prefix, path) if a == b]
if not prefix:
break
return prefix
return filtered, df
common_prefix = longest_common_prefix(split_parts)
prefix_str = os.sep.join(common_prefix)
for i, parts in enumerate(split_parts):
filename = parts[-1]
middle = os.sep.join(parts[len(common_prefix):-1]) if len(parts) > len(common_prefix) + 1 else ""
row = group_df.iloc[i].copy()
row["longestcfp"] = prefix_str
row["middle"] = middle
row["filename_only"] = filename
new_rows.append(row)
return pd.DataFrame(new_rows).drop(columns=["group_key"])
def mask_from_csv(df, csv_path, filepath_col):
"""
@@ -142,11 +127,58 @@ def inspect_parquet(path):
def regulator(paths, case_insensitive=True):
"""
Build a Python raw string regex that matches any of the given Windows path fragments.
Build a regex pattern that matches any of the given Windows path fragments.
"""
escaped = [re.escape(p) for p in paths]
pattern = "(?:" + "|".join(escaped) + ")"
if case_insensitive:
pattern = pattern
print(f"Regulator is providing {pattern}")
return f'r"{pattern}"'
pattern = "(?i)" + pattern # Add inline case-insensitive flag
print(f"Regulator is providing: {pattern}")
return pattern
def generatePathReview(first_policy, second_policy, badpathparts, min_files_for_path):
if not os.path.exists(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet"):
df1 = tryToReadCSV(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv")
df2 = tryToReadCSV(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv")
all_approved_hashes = pd.concat([df1 , df2], ignore_index=True).sort_values(by=['filename'])
print(ct.colorText(f"Approved hash lists have been combined","green"))
all_approved_hashes.to_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet", index=False)
del all_approved_hashes
gc.collect()
if not os.path.exists(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet"):
all_approved_hashes = pd.read_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet")
print(ct.colorText(f"Beginning calculating longest common filepaths for path exceptions","green"))
haslcp = pathf.split_filepaths_grouped(all_approved_hashes)
haslcp.drop_duplicates()
forbidden = pathf.regulator(badpathparts, True)
forbidden_lcfp = haslcp["longestcfp"].str.contains(forbidden, na=False)
print(ct.colorText("Removing forbidden filepaths for path exceptions", "green"))
# Make a real DataFrame copy before modifying
lcp_not_forbidden = haslcp[~forbidden_lcfp].copy()
#For the review, drop down to only the columns we care, and then group by the commmon file path, consolidating and dropping dupes
lcp_not_forbidden_review = lcp_not_forbidden[['longestcfp', 'middle', 'filename_only', 'sha256']]
# Count unique sha256 per longestcfp
unique_sha_counts = lcp_not_forbidden_review.groupby('longestcfp')['sha256'].nunique().reset_index()
unique_sha_counts.columns = ['longestcfp', 'unique_sha256_count']
# Merge the count back into the original DataFrame
lcp_not_forbidden_review = lcp_not_forbidden_review.merge(unique_sha_counts, on='longestcfp', how='left')
lcp_not_forbidden_review = lcp_not_forbidden_review[lcp_not_forbidden_review['unique_sha256_count'] >= min_files_for_path]
lcp_not_forbidden_review.to_parquet(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet",index=False)
lcp_not_forbidden_review.to_csv(f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.csv",index=False)
+74 -15
View File
@@ -12,19 +12,28 @@
#
# You should have received a copy of the GNU Affero General Public License
# along with this program. If not, see <https://www.gnu.org/licenses/>.
import requests
import gc
import json
import os
import pandas as pd
import re
import requests
import utils.pretty as ct
import utils.allowlist
def addHash(policy, hash):
print(f"Adding the following hashes to {policy}:")
print(hash)
for p in hash:
print(p)
def addPath(policy, hash):
print(f"Adding the following Path Exclusions to {policy}:")
print(hash)
for p in hash:
print(p)
def addHashReal(url, allowlistID, hashlist):
endpoint = url + '/v1/hash/application/add'
@@ -36,29 +45,79 @@ def addHashReal(url, allowlistID, hashlist):
headers = {
"X-APIKey": os.getenv('APIKEY')
}
try:
payload = json.dumps(payload)
response = requests.request("POST", endpoint, headers=headers, data=payload, verify=False)
response.raise_for_status() # Raise an error for bad status codes
parse_text = json.loads(response.text)
print(parse_text)
except requests.exceptions.RequestException as e:
return {"error": str(e)}
def addPathReal(url, grouplistID, pathlist):
endpoint = url + '/v1/group/path/add'
print(ct.colorText("[+] Grabbing All Categories", "cyan"))
payload = {
"applicationid" : grouplistID,
"hashes" : pathlist
"groupid" : grouplistID,
"path" : pathlist
}
headers = {
"X-APIKey": os.getenv('APIKEY')
}
try:
print(payload)
payload = json.dumps(payload)
response = requests.request("POST", endpoint, headers=headers, data=payload, verify=False)
response.raise_for_status() # Raise an error for bad status codes
parse_text = json.loads(response.text)
print(parse_text)
except requests.exceptions.RequestException as e:
return {"error": str(e)}
print(response.text)
def getPolicyInfo(url, policy, days):
executionhist_policy = pd.DataFrame()
exehist = utils.allowlist.pullPolicyExechistories(url, policy, days, True)
data = json.loads(exehist)
executionhist_policy = pd.DataFrame(data["response"]["exechistories"])
if not executionhist_policy.empty:
executionhist_policyxecutionhist_policy = executionhist_policy[['sha256', 'publisher', 'filename', 'hostname', 'username', 'pprocess', 'gprocess', 'commandline']]
executionhist_policy = executionhist_policy.drop_duplicates(subset=['sha256', 'filename', 'hostname'])
executionhist_policy = executionhist_policy.sort_values(by=['sha256', 'filename'])
executionhist_policy.to_parquet(f"parquet\\execution_history_{policy}.parquet", index=False)
print(ct.colorText(f"Staging of Execution history for policy: {policy} is complete", "green"))
del data
del exehist
gc.collect()
return executionhist_policy
def sendToPolicy(url, first_policy, second_policy, destination_name, destination_id, allowlist_parent_name, allowlist_parent_id, allowlist_child_name, allowlist_child_id):
pathexclusions = pd.read_parquet(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet")
allowbyhash = pd.read_parquet(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet")
ct.areYouSure()
confirmation = input(ct.colorText("Type 'I AGREE' to continue: ","white"))
if confirmation.strip().upper() == "I AGREE":
print(ct.colorText("Proceeding with the code...", "yellow"))
print(ct.colorText(f"Adding path exclusions to {destination_name}", "yellow"))
pathexcludelist = pathexclusions['longestcfp'].unique().tolist()
# Regex to match a Windows drive letter at the start (e.g., C:\)
drive_letter_pattern = re.compile(r'^[a-zA-Z]:\\')
# Processed list
processed_paths = [
(path if drive_letter_pattern.match(path) else f"\\\\{path}") + "**"
for path in pathexcludelist
]
addPath(url, destination_id,processed_paths)
print(ct.colorText(f"Adding hashes to {allowlist_parent_name}", "yellow"))
allowlist_parenthashlist = allowbyhash[allowbyhash['reputation_status'] == 'KNOWN']['sha256'].unique().tolist()
addHash(url, allowlist_parent_id,allowlist_parenthashlist)
print(ct.colorText(f"Adding hashes to {allowlist_child_name}", "yellow"))
allowlist_childhashlist = allowbyhash[allowbyhash['reputation_status'] == 'UNKNOWN']['sha256'].unique().tolist()
addHash(url, allowlist_child_id, allowlist_childhashlist)
ct.locked()
exit()
else:
print(ct.colorText("Operation aborted. You MUST EXPLICITLY AGREE to proceed.", "red"))
+126
View File
@@ -1,3 +1,18 @@
# Copyright (C) 2025 James Brotosky, Brandon Wickline
#
# This program is free software: you can redistribute it and/or modify
# it under the terms of the GNU Affero General Public License as published
# by the Free Software Foundation, either version 3 of the License, or
# (at your option) any later version.
#
# This program is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# GNU Affero General Public License for more details.
#
# You should have received a copy of the GNU Affero General Public License
# along with this program. If not, see <https://www.gnu.org/licenses/>.
import os
def colorText(text: str, color: str) -> str:
colors = {
@@ -176,6 +191,117 @@ def displayIntro():
print(colorText("======================== Welcome to the Airlock API Tool ========================", "cyan"))
print(colorText("=================================================================================", "cyan"))
def printEnforceChecklist(first_policy, second_policy, allowlist_child_name, allowlist_parent_name, destination_name):
print(colorText("\n --------------------------------------------------------------------", "cyan"))
print(colorText(" -------------------- Prepare to Enforce Policy ---------------------", "cyan"))
print(colorText(" --------------------------------------------------------------------", "cyan"))
print(colorText("\nSequentually follow these steps to prepare a policy for enforcement:", "white"))
print(colorText("\n1. Choose which originating policy or policies to move to enforcement", "cyan"))
if first_policy == " " and second_policy == " ":
print(colorText(f" [✗] No policies have been chosen","red"))
elif first_policy != " " and second_policy is first_policy:
print(colorText(f" [✓] {first_policy} has been selected,", "green"))
elif first_policy != " " and second_policy != " ":
print(colorText(f" [✓] {first_policy} has been selected as Policy 1","green"))
print(colorText(f" [✓] {second_policy} has been selected as Policy 2","green"))
print(colorText("2. Pull and stage event history, combine the histories, add hash info, then categorize the hashes", "cyan"))
if os.path.exists(f"parquet\\execution_history_{first_policy}.parquet"):
print(colorText(f" [✓] Execution history has been compiled for {first_policy}","green"))
elif not os.path.exists(f"parquet\\execution_history_{first_policy}.parquet"):
print(colorText(f" [✗] Execution history has not been compiled for {first_policy}","red"))
elif second_policy is not first_policy and os.path.exists(f"parquet\\execution_history_{second_policy}.parquet"):
print(colorText(f" [✓] Execution history has been compiled for {second_policy}","green"))
elif second_policy is not first_policy and not os.path.exists(f"parquet\\execution_history_{second_policy}.parquet"):
print(colorText(f" [✗] Execution history has not been compiled for {second_policy}","red"))
if os.path.exists(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"):
print(colorText(f" [✓] Hash Info has been added to the combined execution history", "green"))
else:
print(colorText(f" [✗] Hash Info has not been added to the combined execution history", "red"))
if os.path.exists(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") and os.path.exists(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") and os.path.exists(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet"):
print(colorText(f" [✓] Hashes have been cateogrized", "green"))
else:
print(colorText(f" [✗] Hashes have not been cateogrized", "red"))
if os.path.exists(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet"):
print(colorText(f" [✓] Execution history has been_combined_for {first_policy} and_{second_policy}", "green"))
else:
print(colorText(f" [✗] Execution history has not been_combined_for {first_policy} and_{second_policy}", "red"))
print(colorText(f"3. Manually review the files:","cyan"))
print(colorText(" 'needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv' and 'needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv'", "cyan"))
print(colorText(" Remove the rows containing hashes you do not approve of, and those you would not approve of without metarules.", "cyan"))
print(colorText(" If metarules need to be created, please make note of them, and remove the row from the csv.", "cyan"))
print(colorText(" When complete, save both csv files to the directory 'approved' and choose this option.","cyan"))
print(colorText(" This will combine these approved hashes with the automatically approved hashes and generate a list of paths to be reviewed", "cyan"))
if os.path.exists(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv") and os.path.exists(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv"):
print(colorText(" [✓] Reviewed hashes have been loaded","green"))
else:
print(colorText(" [✗] Reviewed hashes have not been loaded","red"))
if os.path.exists(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet"):
print(colorText(" [✓] The combined approved hashes list has been generated","green"))
else:
print(colorText(" [✗] The combined approved hashes list has not been generated","red"))
if os.path.exists(f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.csv"):
print(colorText(" [✓] Path review list created","green"))
else:
print(colorText(" [✗] Path review list has not been created","red"))
print(colorText(f"4. Manually review the file 'needs_approved\\paths_needing_review_{first_policy}_{second_policy}.csv'", "cyan"))
print(colorText(" Remove the rows containing path exclusions you do not approve of" , "cyan"))
print(colorText(" When complete, save the csv file to the directory 'approved'", "cyan"))
print(colorText(" Preflight Lists will be generated", "cyan"))
if os.path.exists(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv"):
print(colorText(" [✓] Reviewed path list detected","green"))
else:
print(colorText(" [✗] Path review list has not been detected","red"))
if os.path.exists(f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html"):
print(colorText(" [✓] Preflight Path Exclusion List has been generated","green"))
else:
print(colorText(" [✗] Preflight Path Exclusion List has not been generated","red"))
if os.path.exists(f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html"):
print(colorText(" [✓] Preflight hash approval list has been generated","green"))
else:
print(colorText(" [✗] Preflight hash approval list has not been generated","red"))
print(colorText(f"5. Choose the destination policy and parent and child allow list", "cyan"))
if allowlist_child_name == " " and allowlist_parent_name== " ":
print(colorText(f" [✗] No allowlists have been chosen","red"))
elif allowlist_parent_name != " " and allowlist_child_name != " " and allowlist_parent_name is allowlist_child_name:
print(colorText(f" [✓] [✗] Only {allowlist_parent_name} has been selected this is unusual, but potentially valid case, double check before proceeding,", "yellow"))
elif allowlist_parent_name != " " and allowlist_child_name != " " and allowlist_parent_name is not allowlist_child_name:
print(colorText(f" [✓] {allowlist_parent_name} has been selected as Parent Policy","green"))
print(colorText(f" [✓] {allowlist_child_name} has been selected as Child Policy","green"))
if destination_name == " ":
print(colorText(f" [✗] No destination policy has been chosen","red"))
else:
print(colorText(f" [✓] destination policy is {destination_name}","green"))
print(colorText(f"6. Liftoff ------------------------------------------------------", "cyan"))
print(colorText(f" Apply path exclusions according to allowed and approved paths", "cyan"))
print(colorText(f" Apply signed or attested hashes to Parent Allow List", "cyan"))
print(colorText(f" Apply approved, but unsigned hashes to the Child Allow List", "cyan"))
print(colorText("Q. Quit", "cyan"))
def areYouSure():
print(colorText(f"*******************************************************************************************************************************************","red"))