IOTesting #19
+36
-417
@@ -14,8 +14,6 @@
|
|||||||
# along with this program. If not, see <https://www.gnu.org/licenses/>.
|
# along with this program. If not, see <https://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
import dotenv
|
import dotenv
|
||||||
import gc
|
|
||||||
import json
|
|
||||||
import os
|
import os
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
import urllib3
|
import urllib3
|
||||||
@@ -26,19 +24,20 @@ import utils.pathfunctions
|
|||||||
import utils.policyfunctions
|
import utils.policyfunctions
|
||||||
import utils.pretty as ct
|
import utils.pretty as ct
|
||||||
|
|
||||||
|
|
||||||
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
|
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
|
||||||
|
|
||||||
dotenv.load_dotenv()
|
dotenv.load_dotenv()
|
||||||
|
|
||||||
#Constants
|
#Constants
|
||||||
url = os.getenv('url')
|
url = os.getenv('url')
|
||||||
badpublisherlist = ["Brave Software, Inc.", "Zoom Video Communications, Inc."]
|
bad_publisher_list = ["Brave","Zoom", "GlavSoft", "VNC"]
|
||||||
badpathparts = ["users", "inet\\wwwroot", "windows\\temp", "windows\\temp", "windows\\task", "windows\\system32", "startup", "windows\\fonts", "Recycle.Bin", "AppData", "programdata"]
|
pups = ["logmein", "invalid"]
|
||||||
path_exclusion_constant = 3
|
badpathparts = ["users", "wwwroot", "windows\\temp", "windows\\task", "windows\\system32", "startup", "windows\\fonts", "Recycle.Bin", "AppData", "programdata"]
|
||||||
|
path_exclusion_constant = 4
|
||||||
min_files_for_path = 4
|
min_files_for_path = 4
|
||||||
threat_tolerance_constant = 4
|
threat_tolerance_constant = 4
|
||||||
|
|
||||||
|
|
||||||
def apivalidation():
|
def apivalidation():
|
||||||
match os.getenv('APIKEY'):
|
match os.getenv('APIKEY'):
|
||||||
case '':
|
case '':
|
||||||
@@ -137,10 +136,12 @@ def menu_prepare_to_enforce():
|
|||||||
|
|
||||||
first_policy = " "
|
first_policy = " "
|
||||||
second_policy = " "
|
second_policy = " "
|
||||||
allowlist_parent_name = " "
|
|
||||||
allowlist_child_name = " "
|
|
||||||
destination_name = " "
|
destination_name = " "
|
||||||
df_aggregated_combo = pd.DataFrame()
|
destination_id = " "
|
||||||
|
allowlist_parent_name = " "
|
||||||
|
allowlist_parent_id = " "
|
||||||
|
allowlist_child_name = " "
|
||||||
|
allowlist_child_id = " "
|
||||||
|
|
||||||
#If the directorys where we're going to store our output dont exist, make them.
|
#If the directorys where we're going to store our output dont exist, make them.
|
||||||
if not os.path.exists("parquet"): os.makedirs("parquet")
|
if not os.path.exists("parquet"): os.makedirs("parquet")
|
||||||
@@ -149,119 +150,8 @@ def menu_prepare_to_enforce():
|
|||||||
if not os.path.exists("preflight"): os.makedirs("preflight")
|
if not os.path.exists("preflight"): os.makedirs("preflight")
|
||||||
|
|
||||||
while True:
|
while True:
|
||||||
print(ct.colorText("\n --------------------------------------------------------------------", "cyan"))
|
|
||||||
print(ct.colorText(" -------------------- Prepare to Enforce Policy ---------------------", "cyan"))
|
|
||||||
print(ct.colorText(" --------------------------------------------------------------------", "cyan"))
|
|
||||||
print(ct.colorText("\nSequentually follow these steps to prepare a policy for enforcement:", "white"))
|
|
||||||
|
|
||||||
|
ct.printEnforceChecklist(first_policy, second_policy, allowlist_child_name, allowlist_parent_name, destination_name)
|
||||||
print(ct.colorText("\n1. Choose which originating policy or policies to move to enforcement", "cyan"))
|
|
||||||
|
|
||||||
if first_policy == " " and second_policy == " ":
|
|
||||||
print(ct.colorText(f" [✗] No policies have been chosen","red"))
|
|
||||||
elif first_policy != " " and second_policy is first_policy:
|
|
||||||
print(ct.colorText(f" [✓] {first_policy} has been selected,", "green"))
|
|
||||||
elif first_policy != " " and second_policy != " ":
|
|
||||||
print(ct.colorText(f" [✓] {first_policy} has been selected as Policy 1","green"))
|
|
||||||
print(ct.colorText(f" [✓] {second_policy} has been selected as Policy 2","green"))
|
|
||||||
|
|
||||||
|
|
||||||
print(ct.colorText("2. Pull and stage event history, combine the histories, add hash info, then categorize the hashes", "cyan"))
|
|
||||||
|
|
||||||
if os.path.exists(f"parquet\\execution_history_{first_policy}.parquet"):
|
|
||||||
print(ct.colorText(f" [✓] Execution history has been compiled for {first_policy}","green"))
|
|
||||||
elif not os.path.exists(f"parquet\\execution_history_{first_policy}.parquet"):
|
|
||||||
print(ct.colorText(f" [✗] Execution history has not been compiled for {first_policy}","red"))
|
|
||||||
elif second_policy is not first_policy and os.path.exists(f"parquet\\execution_history_{second_policy}.parquet"):
|
|
||||||
print(ct.colorText(f" [✓] Execution history has been compiled for {second_policy}","green"))
|
|
||||||
elif second_policy is not first_policy and not os.path.exists(f"parquet\\execution_history_{second_policy}.parquet"):
|
|
||||||
print(ct.colorText(f" [✗] Execution history has not been compiled for {second_policy}","red"))
|
|
||||||
|
|
||||||
if os.path.exists(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"):
|
|
||||||
print(ct.colorText(f" [✓] Hash Info has been added to the combined execution history", "green"))
|
|
||||||
else:
|
|
||||||
print(ct.colorText(f" [✗] Hash Info has not been added to the combined execution history", "red"))
|
|
||||||
|
|
||||||
if os.path.exists(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") and os.path.exists(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") and os.path.exists(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet"):
|
|
||||||
print(ct.colorText(f" [✓] Hashes have been cateogrized", "green"))
|
|
||||||
else:
|
|
||||||
print(ct.colorText(f" [✗] Hashes have not been cateogrized", "red"))
|
|
||||||
|
|
||||||
if os.path.exists(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet"):
|
|
||||||
print(ct.colorText(f" [✓] Execution history has been_combined_for {first_policy} and_{second_policy}", "green"))
|
|
||||||
else:
|
|
||||||
print(ct.colorText(f" [✗] Execution history has not been_combined_for {first_policy} and_{second_policy}", "red"))
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
print(ct.colorText(f"3. Manually review the files:","cyan"))
|
|
||||||
print(ct.colorText(" 'needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv' and 'needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv'", "cyan"))
|
|
||||||
print(ct.colorText(" Remove the rows containing hashes you do not approve of, and those you would not approve of without metarules.", "cyan"))
|
|
||||||
print(ct.colorText(" If metarules need to be created, please make note of them, and remove the row from the csv.", "cyan"))
|
|
||||||
print(ct.colorText(" When complete, save both csv files to the directory 'approved' and choose this option.","cyan"))
|
|
||||||
print(ct.colorText(" This will combine these approved hashes with the automatically approved hashes and generate a list of paths to be reviewed", "cyan"))
|
|
||||||
|
|
||||||
if os.path.exists(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv") and os.path.exists(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv"):
|
|
||||||
print(ct.colorText(" [✓] Reviewed hashes have been loaded","green"))
|
|
||||||
else:
|
|
||||||
print(ct.colorText(" [✗] Reviewed hashes have not been loaded","red"))
|
|
||||||
|
|
||||||
if os.path.exists(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet"):
|
|
||||||
print(ct.colorText(" [✓] The combined approved hashes list has been generated","green"))
|
|
||||||
else:
|
|
||||||
print(ct.colorText(" [✗] The combined approved hashes list has not been generated","red"))
|
|
||||||
|
|
||||||
if os.path.exists(f"parquet\\approved_hashes_with_paths_{first_policy}_{second_policy}.parquet"):
|
|
||||||
print(ct.colorText(" [✓] Longest common filepaths have been generated and appended to hash info","green"))
|
|
||||||
else:
|
|
||||||
print(ct.colorText(" [✗] Longest common filepaths have not been generated","red"))
|
|
||||||
|
|
||||||
if os.path.exists(f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.csv"):
|
|
||||||
print(ct.colorText(" [✓] Path review list created","green"))
|
|
||||||
else:
|
|
||||||
print(ct.colorText(" [✗] Path review list has not been created","red"))
|
|
||||||
|
|
||||||
|
|
||||||
print(ct.colorText(f"4. Manually review the file 'needs_approved\\paths_needing_review_{first_policy}_{second_policy}.csv'", "cyan"))
|
|
||||||
print(ct.colorText(" Remove the rows containing path exclusions you do not approve of" , "cyan"))
|
|
||||||
print(ct.colorText(" When complete, save the csv file to the directory 'approved'", "cyan"))
|
|
||||||
print(ct.colorText(" Preflight Lists will be generated", "cyan"))
|
|
||||||
|
|
||||||
if os.path.exists(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv"):
|
|
||||||
print(ct.colorText(" [✓] Reviewed path list detected","green"))
|
|
||||||
else:
|
|
||||||
print(ct.colorText(" [✗] Path review list has not been detected","red"))
|
|
||||||
|
|
||||||
if os.path.exists(f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html"):
|
|
||||||
print(ct.colorText(" [✓] Preflight Path Exclusion List has been generated","green"))
|
|
||||||
else:
|
|
||||||
print(ct.colorText(" [✗] Preflight Path Exclusion List has not been generated","red"))
|
|
||||||
|
|
||||||
if os.path.exists(f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html"):
|
|
||||||
print(ct.colorText(" [✓] Preflight hash approval list has been generated","green"))
|
|
||||||
else:
|
|
||||||
print(ct.colorText(" [✗] Preflight hash approval list has not been generated","red"))
|
|
||||||
|
|
||||||
|
|
||||||
print(ct.colorText(f"5. Choose the destination policy and parent and child allow list", "cyan"))
|
|
||||||
if allowlist_child_name == " " and allowlist_parent_name== " ":
|
|
||||||
print(ct.colorText(f" [✗] No allowlists have been chosen","red"))
|
|
||||||
elif allowlist_parent_name != " " and allowlist_child_name != " " and allowlist_parent_name is allowlist_child_name:
|
|
||||||
print(ct.colorText(f" [✓] [✗] Only {allowlist_parent_name} has been selected this is unusual, but potentially valid case, double check before proceeding,", "yellow"))
|
|
||||||
elif allowlist_parent_name != " " and allowlist_child_name != " " and allowlist_parent_name is not allowlist_child_name:
|
|
||||||
print(ct.colorText(f" [✓] {allowlist_parent_name} has been selected as Parent Policy","green"))
|
|
||||||
print(ct.colorText(f" [✓] {allowlist_child_name} has been selected as Child Policy","green"))
|
|
||||||
if destination_name == " ":
|
|
||||||
print(ct.colorText(f" [✗] No destination policy has been chosen","red"))
|
|
||||||
else:
|
|
||||||
print(ct.colorText(f" [✓] destination policy is {destination_name}","green"))
|
|
||||||
|
|
||||||
print(ct.colorText(f"6. Liftoff ------------------------------------------------------", "cyan"))
|
|
||||||
print(ct.colorText(f" Apply path exclusions according to allowed and approved paths", "cyan"))
|
|
||||||
print(ct.colorText(f" Apply signed or attested hashes to Parent Allow List", "cyan"))
|
|
||||||
print(ct.colorText(f" Apply approved, but unsigned hashes to the Child Allow List", "cyan"))
|
|
||||||
|
|
||||||
print(ct.colorText("Q. Quit", "cyan"))
|
|
||||||
|
|
||||||
choice = input(ct.colorText("\nEnter your choice: ", "white"))
|
choice = input(ct.colorText("\nEnter your choice: ", "white"))
|
||||||
|
|
||||||
@@ -285,296 +175,42 @@ def menu_prepare_to_enforce():
|
|||||||
elif choice == "2":
|
elif choice == "2":
|
||||||
|
|
||||||
if not os.path.exists(f"parquet\\execution_history_{first_policy}.parquet"):
|
if not os.path.exists(f"parquet\\execution_history_{first_policy}.parquet"):
|
||||||
print(choice)
|
utils.policyfunctions.getPolicyInfo(url, first_policy, 60)
|
||||||
print(first_policy)
|
|
||||||
|
|
||||||
exe1 = utils.allowlist.pullPolicyExechistories(url, first_policy, 60, True)
|
|
||||||
data = json.loads(exe1)
|
|
||||||
executionhist_policy1 = pd.DataFrame(data["response"]["exechistories"])
|
|
||||||
|
|
||||||
if not executionhist_policy1.empty:
|
|
||||||
executionhist_policy1 = executionhist_policy1[['sha256', 'publisher', 'filename', 'hostname', 'username', 'pprocess', 'gprocess', 'commandline']]
|
|
||||||
executionhist_policy1 = executionhist_policy1.drop_duplicates(subset=['sha256', 'filename', 'hostname'])
|
|
||||||
executionhist_policy1 = executionhist_policy1.sort_values(by=['sha256', 'filename'])
|
|
||||||
|
|
||||||
executionhist_policy1.to_parquet(f"parquet\\execution_history_{first_policy}.parquet", index=False)
|
|
||||||
print(ct.colorText(f"Staging of Execution history for policy: {first_policy} is complete", "green"))
|
|
||||||
|
|
||||||
del data
|
|
||||||
del exe1
|
|
||||||
del executionhist_policy1
|
|
||||||
|
|
||||||
gc.collect()
|
|
||||||
|
|
||||||
if not os.path.exists(f"parquet\\execution_history_{second_policy}.parquet"):
|
if not os.path.exists(f"parquet\\execution_history_{second_policy}.parquet"):
|
||||||
exe2 = utils.allowlist.pullPolicyExechistories(url,second_policy, 60, True)
|
utils.policyfunctions.getPolicyInfo(url, second_policy, 60)
|
||||||
data2 = json.loads(exe2)
|
|
||||||
executionhist_policy2 = pd.DataFrame(data2["response"]["exechistories"])
|
|
||||||
|
|
||||||
if not executionhist_policy2.empty:
|
|
||||||
executionhist_policy2 = executionhist_policy2[['sha256', 'publisher', 'filename', 'hostname', 'username', 'pprocess', 'gprocess', 'commandline']]
|
|
||||||
executionhist_policy2 = executionhist_policy2.drop_duplicates(subset=['sha256', 'filename', 'hostname'])
|
|
||||||
executionhist_policy2 = executionhist_policy2.sort_values(by=['sha256', 'filename'])
|
|
||||||
|
|
||||||
executionhist_policy2.to_parquet(f"parquet\\execution_history_{second_policy}.parquet", index=False)
|
|
||||||
print(ct.colorText(f"Staging of Execution history for policy: {second_policy} is complete", "green"))
|
|
||||||
|
|
||||||
del executionhist_policy2
|
|
||||||
del data2
|
|
||||||
del exe2
|
|
||||||
|
|
||||||
gc.collect()
|
|
||||||
|
|
||||||
if not os.path.exists(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"):
|
if not os.path.exists(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"):
|
||||||
combined_hashes = pd.DataFrame(columns=['sha256', 'publisher'])
|
utils.hashfunctions.combineHashes(url, first_policy, second_policy)
|
||||||
hashes = []
|
|
||||||
|
|
||||||
try:
|
|
||||||
hash1 = pd.read_parquet(f"parquet\\execution_history_{first_policy}.parquet", columns=['sha256', 'publisher'])
|
|
||||||
utils.pathfunctions.inspect_parquet(f"parquet\\execution_history_{first_policy}.parquet")
|
|
||||||
if not hash1.empty:
|
|
||||||
hashes.append(hash1)
|
|
||||||
else:
|
|
||||||
print("⚠️ First dataframe is empty.")
|
|
||||||
except Exception as e:
|
|
||||||
print(f"❌ Error reading first Parquet file: {e}")
|
|
||||||
|
|
||||||
try:
|
|
||||||
hash2 = pd.read_parquet(f"parquet\\execution_history_{second_policy}.parquet", columns=['sha256', 'publisher'])
|
|
||||||
utils.pathfunctions.inspect_parquet(f"parquet\\execution_history_{second_policy}.parquet")
|
|
||||||
if not hash2.empty:
|
|
||||||
hashes.append(hash2)
|
|
||||||
else:
|
|
||||||
print("⚠️ Second dataframe is empty.")
|
|
||||||
except Exception as e:
|
|
||||||
print(f"❌ Error reading second Parquet file: {e}")
|
|
||||||
|
|
||||||
if hashes:
|
|
||||||
combined_hashes = pd.concat(hashes, ignore_index=True)
|
|
||||||
print(f"✅ Combined {len(combined_hashes)} hashes.")
|
|
||||||
else:
|
|
||||||
print("⚠️ No valid dataframes to combine.")
|
|
||||||
|
|
||||||
combined_hashes = combined_hashes.drop_duplicates(subset=['sha256'])
|
|
||||||
augmented_combo = utils.hashfunctions.augmentAggregatedHashes(url, combined_hashes)
|
|
||||||
|
|
||||||
numeric_reputation_cols = [
|
|
||||||
'reputation_scannermatch',
|
|
||||||
'reputation_scannercount',
|
|
||||||
'reputation_threatlevel'
|
|
||||||
]
|
|
||||||
|
|
||||||
for col in numeric_reputation_cols:
|
|
||||||
if col in augmented_combo.columns:
|
|
||||||
augmented_combo[col] = pd.to_numeric(augmented_combo[col].replace('N/A', pd.NA), errors='coerce')
|
|
||||||
|
|
||||||
augmented_combo = augmented_combo.rename(columns={'publisher_x': 'publisher'})
|
|
||||||
augmented_combo = augmented_combo[['sha256', 'publisher', 'description', 'productname', 'productversion',
|
|
||||||
'reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount',
|
|
||||||
'reputation_status', 'reputation_threatlevel', 'reputation_threatname',
|
|
||||||
'reputation_timestamp']]
|
|
||||||
augmented_combo = augmented_combo.sort_values(by=['publisher', 'description', 'productname'])
|
|
||||||
augmented_combo.to_parquet(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet", index=False)
|
|
||||||
|
|
||||||
del combined_hashes
|
|
||||||
del augmented_combo
|
|
||||||
gc.collect()
|
|
||||||
|
|
||||||
print(ct.colorText("Hash reputation info added to dataframe", "green"))
|
|
||||||
|
|
||||||
if not os.path.exists(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet"):
|
if not os.path.exists(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet"):
|
||||||
|
utils.hashfunctions.categorizeHashes(
|
||||||
# Categorize the hashes
|
first_policy,
|
||||||
categorized = utils.hashfunctions.categorizeHashes(
|
second_policy,
|
||||||
pd.read_parquet(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"),
|
pd.read_parquet(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"),
|
||||||
threat_tolerance_constant,
|
threat_tolerance_constant,
|
||||||
badpublisherlist
|
bad_publisher_list,
|
||||||
|
pups
|
||||||
)
|
)
|
||||||
|
|
||||||
categorized[0].to_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet", index=False)
|
|
||||||
categorized[1].to_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", index=False)
|
|
||||||
categorized[2].to_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", index=False)
|
|
||||||
|
|
||||||
del categorized
|
|
||||||
gc.collect()
|
|
||||||
|
|
||||||
if not os.path.exists(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet"):
|
if not os.path.exists(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet"):
|
||||||
# Condense execution history
|
utils.hashfunctions.condenseExecutions(first_policy,second_policy)
|
||||||
try:
|
|
||||||
exe1 = pd.read_parquet(f"parquet\\execution_history_{first_policy}.parquet")
|
|
||||||
utils.pathfunctions.inspect_parquet(f"parquet\\execution_history_{first_policy}.parquet")
|
|
||||||
if not exe1.empty:
|
|
||||||
condensed_exe1 = exe1.groupby('sha256').agg(lambda x: list(set(x))).reset_index()
|
|
||||||
else:
|
|
||||||
print("⚠️ First dataframe is empty.")
|
|
||||||
except Exception as e:
|
|
||||||
print(f"❌ Error reading first Parquet file: {e}")
|
|
||||||
|
|
||||||
try:
|
if os.path.exists(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") & os.path.exists(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") & os.path.exists(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet"):
|
||||||
exe2 = pd.read_parquet(f"parquet\\execution_history_{second_policy}.parquet")
|
utils.hashfunctions.divideSortedHashExecutions(first_policy,second_policy,pups)
|
||||||
utils.pathfunctions.inspect_parquet(f"parquet\\execution_history_{second_policy}.parquet")
|
|
||||||
if not exe2.empty:
|
|
||||||
condensed_exe2 = exe2.groupby('sha256').agg(lambda x: list(set(x))).reset_index()
|
|
||||||
else:
|
|
||||||
print("⚠️ Second dataframe is empty.")
|
|
||||||
except Exception as e:
|
|
||||||
print(f"❌ Error reading second Parquet file: {e}")
|
|
||||||
|
|
||||||
if not exe1.empty and not exe2.empty:
|
|
||||||
condensed_combo = pd.concat([condensed_exe1, condensed_exe2], ignore_index=True)
|
|
||||||
print(f"✅ Combined {len(condensed_combo)} hashes.")
|
|
||||||
elif exe1.empty:
|
|
||||||
condensed_combo = condensed_exe2
|
|
||||||
elif exe2.empty:
|
|
||||||
condensed_combo = condensed_exe1
|
|
||||||
else:
|
|
||||||
print("⚠️ No valid dataframes to combine.")
|
|
||||||
|
|
||||||
condensed_combo.to_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet", index=False)
|
|
||||||
del condensed_combo
|
|
||||||
gc.collect()
|
|
||||||
|
|
||||||
if not os.path.exists(f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv"):
|
|
||||||
|
|
||||||
condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet")
|
|
||||||
needsapproval= pd.read_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet")
|
|
||||||
|
|
||||||
#Pull hash info for the entries in the needs approval table
|
|
||||||
needsapproval = pd.merge(condensed_combo, needsapproval, on='sha256', how='inner')
|
|
||||||
|
|
||||||
#Deduplicate lists in the columns
|
|
||||||
for col in needsapproval.columns:
|
|
||||||
if needsapproval[col].apply(lambda x: isinstance(x, list)).all():
|
|
||||||
needsapproval[col] = needsapproval[col].apply(deduplicate_list)
|
|
||||||
|
|
||||||
#Rename Publisher, Keep and reorder columns we want
|
|
||||||
needsapproval = needsapproval.rename(columns={'publisher_x': 'publisher'})
|
|
||||||
needsapproval = needsapproval[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']]
|
|
||||||
|
|
||||||
needsapproval.to_csv(f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv",index=False)
|
|
||||||
needsapproval.to_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet",index=False)
|
|
||||||
ct.style_dataframe_dark(needsapproval, f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.html")
|
|
||||||
|
|
||||||
|
|
||||||
del needsapproval
|
|
||||||
del condensed_combo
|
|
||||||
gc.collect()
|
|
||||||
|
|
||||||
if not os.path.exists(f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv"):
|
|
||||||
|
|
||||||
condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet")
|
|
||||||
needsapproval= pd.read_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet")
|
|
||||||
|
|
||||||
#Pull hash info for the entries in the needs approval table
|
|
||||||
needsapproval = pd.merge(condensed_combo, needsapproval, on='sha256', how='inner')
|
|
||||||
|
|
||||||
#Deduplicate lists in the columns
|
|
||||||
for col in needsapproval.columns:
|
|
||||||
if needsapproval[col].apply(lambda x: isinstance(x, list)).all():
|
|
||||||
needsapproval[col] = needsapproval[col].apply(deduplicate_list)
|
|
||||||
|
|
||||||
#Rename Publisher, Keep and reorder columns we want
|
|
||||||
needsapproval = needsapproval.rename(columns={'publisher_x': 'publisher'})
|
|
||||||
needsapproval = needsapproval[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']]
|
|
||||||
|
|
||||||
needsapproval.to_csv(f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv",index=False)
|
|
||||||
needsapproval.to_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", index=False)
|
|
||||||
ct.style_dataframe_dark(needsapproval, f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.html")
|
|
||||||
|
|
||||||
|
|
||||||
del needsapproval
|
|
||||||
del condensed_combo
|
|
||||||
gc.collect()
|
|
||||||
|
|
||||||
if not os.path.exists(f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.html"):
|
|
||||||
|
|
||||||
condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet")
|
|
||||||
needsapproval= pd.read_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet")
|
|
||||||
|
|
||||||
#Pull hash info for the entries in the needs approval table
|
|
||||||
needsapproval = pd.merge(condensed_combo, needsapproval, on='sha256', how='inner')
|
|
||||||
|
|
||||||
#Deduplicate lists in the columns
|
|
||||||
for col in needsapproval.columns:
|
|
||||||
if needsapproval[col].apply(lambda x: isinstance(x, list)).all():
|
|
||||||
needsapproval[col] = needsapproval[col].apply(deduplicate_list)
|
|
||||||
|
|
||||||
#Rename Publisher, Keep and reorder columns we want
|
|
||||||
needsapproval = needsapproval.rename(columns={'publisher_x': 'publisher'})
|
|
||||||
needsapproval = needsapproval[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']]
|
|
||||||
|
|
||||||
needsapproval.to_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", index=False)
|
|
||||||
ct.style_dataframe_dark(needsapproval, f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.html")
|
|
||||||
|
|
||||||
del needsapproval
|
|
||||||
del condensed_combo
|
|
||||||
gc.collect()
|
|
||||||
|
|
||||||
elif choice == "3":
|
elif choice == "3":
|
||||||
|
|
||||||
if os.path.exists(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv") and os.path.exists(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv"):
|
if os.path.exists(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv") and os.path.exists(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv"):
|
||||||
|
utils.pathfunctions.generatePathReview(first_policy, second_policy, badpathparts, min_files_for_path)
|
||||||
if not os.path.exists(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet"):
|
|
||||||
|
|
||||||
df1 = tryToReadCSV(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv")
|
|
||||||
df2 = tryToReadCSV(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv")
|
|
||||||
|
|
||||||
all_approved_hashes = pd.concat([df1 , df2], ignore_index=True).sort_values(by=['sha256','filename'])
|
|
||||||
print(ct.colorText(f"Approved hash lists have been combined","green"))
|
|
||||||
|
|
||||||
all_approved_hashes.to_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet", index=False)
|
|
||||||
del all_approved_hashes
|
|
||||||
gc.collect()
|
|
||||||
|
|
||||||
if not os.path.exists(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet"):
|
|
||||||
all_approved_hashes = pd.read_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet")
|
|
||||||
print(ct.colorText(f"Beginning calculating longest common filepaths for path exceptions","green"))
|
|
||||||
grouped_df_view, df_with_groups_appended = utils.pathfunctions.export_groups_for_review(all_approved_hashes,"filename","longestcfp",min_files_for_path,path_exclusion_constant)
|
|
||||||
|
|
||||||
df_with_groups_appended.to_parquet(f"parquet\\approved_hashes_with_paths_{first_policy}_{second_policy}.parquet", index=False)
|
|
||||||
|
|
||||||
forbidden = utils.pathfunctions.regulator(badpathparts, True)
|
|
||||||
forbidden_lcfp = grouped_df_view["longestcfp"].str.contains(forbidden, na=False)
|
|
||||||
grouped_df_view = grouped_df_view[~forbidden_lcfp]
|
|
||||||
print(ct.colorText(f"Removing forbidden filepaths for path exceptions","green"))
|
|
||||||
|
|
||||||
grouped_df_view.to_parquet(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet", index=False)
|
|
||||||
grouped_df_view.to_csv(f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.csv", index=False)
|
|
||||||
ct.style_dataframe_dark(grouped_df_view, f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.html")
|
|
||||||
|
|
||||||
del grouped_df_view
|
|
||||||
del df_with_groups_appended
|
|
||||||
gc.collect()
|
|
||||||
|
|
||||||
else:
|
else:
|
||||||
print(ct.colorText(f"Please manually approve hashes prior to this step","red"))
|
print(ct.colorText(f"Please manually approve hashes prior to this step","red"))
|
||||||
|
|
||||||
|
|
||||||
elif choice == "4":
|
elif choice == "4":
|
||||||
|
|
||||||
if os.path.exists(f"parquet\\approved_hashes_with_paths_{first_policy}_{second_policy}.parquet") and os.path.exists(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv"):
|
if os.path.exists(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv"):
|
||||||
if not os.path.exists(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet"):
|
if not os.path.exists(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet"):
|
||||||
df1 = pd.read_parquet(f"parquet\\approved_hashes_with_paths_{first_policy}_{second_policy}.parquet")
|
utils.hashfunctions.generatePreflights(first_policy, second_policy)
|
||||||
allowbyhash = utils.pathfunctions.mask_from_csv(df1, f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv","longestcfp")
|
|
||||||
|
|
||||||
pathexclusions = tryToReadCSV(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv")
|
|
||||||
|
|
||||||
pathexclusions.to_parquet(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet", index=False)
|
|
||||||
|
|
||||||
allowbyhash.to_parquet(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet", index=False)
|
|
||||||
|
|
||||||
easyview = allowbyhash.groupby('sha256').agg(list).reset_index()
|
|
||||||
|
|
||||||
ct.style_dataframe_dark(easyview, f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html")
|
|
||||||
ct.style_dataframe_dark(pathexclusions, f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html")
|
|
||||||
|
|
||||||
del df1
|
|
||||||
del allowbyhash
|
|
||||||
del pathexclusions
|
|
||||||
del easyview
|
|
||||||
gc.collect()
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
elif choice == "5":
|
elif choice == "5":
|
||||||
|
|
||||||
@@ -598,34 +234,17 @@ def menu_prepare_to_enforce():
|
|||||||
|
|
||||||
elif choice == "6":
|
elif choice == "6":
|
||||||
if os.path.exists(f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html") and os.path.exists(f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html") and allowlist_parent_name != " " and allowlist_child_name != " " and destination_name != " ":
|
if os.path.exists(f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html") and os.path.exists(f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html") and allowlist_parent_name != " " and allowlist_child_name != " " and destination_name != " ":
|
||||||
|
utils.policyfunctions.sendToPolicy(
|
||||||
pathexclusions = pd.read_parquet(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet")
|
url,
|
||||||
allowbyhash = pd.read_parquet(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet")
|
first_policy,
|
||||||
|
second_policy,
|
||||||
ct.areYouSure()
|
destination_name,
|
||||||
confirmation = input(ct.colorText("Type 'I AGREE' to continue: ","white"))
|
destination_id,
|
||||||
|
allowlist_parent_name,
|
||||||
if confirmation.strip().upper() == "I AGREE":
|
allowlist_parent_id,
|
||||||
print(ct.colorText("Proceeding with the code...", "yellow"))
|
allowlist_child_name,
|
||||||
print(ct.colorText(f"Adding path exclusions to {destination_name}", "yellow"))
|
allowlist_child_id
|
||||||
pathexcludelist = pathexclusions['longestcfp'].unique().tolist()
|
)
|
||||||
utils.policyfunctions.addPath(destination_id,pathexcludelist)
|
|
||||||
|
|
||||||
print(ct.colorText(f"Adding hashes to {allowlist_parent_name}", "yellow"))
|
|
||||||
|
|
||||||
allowlist_parenthashlist = allowbyhash[allowbyhash['reputation_status'] == 'KNOWN']['sha256'].unique().tolist()
|
|
||||||
utils.policyfunctions.addHash(allowlist_parent_id,allowlist_parenthashlist)
|
|
||||||
|
|
||||||
print(ct.colorText(f"Adding hashes to {allowlist_child_name}", "yellow"))
|
|
||||||
allowlist_childhashlist = allowbyhash[allowbyhash['reputation_status'] == 'UNKNOWN']['sha256'].unique().tolist()
|
|
||||||
utils.policyfunctions.addHash(allowlist_child_id, allowlist_childhashlist)
|
|
||||||
|
|
||||||
ct.locked()
|
|
||||||
exit()
|
|
||||||
|
|
||||||
else:
|
|
||||||
print(ct.colorText("Operation aborted. You MUST EXPLICITLY AGREE to proceed.", "red"))
|
|
||||||
break
|
|
||||||
|
|
||||||
elif choice == "Q":
|
elif choice == "Q":
|
||||||
break
|
break
|
||||||
|
|||||||
+27
-2
@@ -1,4 +1,29 @@
|
|||||||
pandas==2.3.2
|
bson==0.5.10
|
||||||
|
certifi==2025.8.3
|
||||||
|
charset-normalizer==3.4.3
|
||||||
|
colorama==0.4.6
|
||||||
|
cramjam==2.11.0
|
||||||
|
docopt==0.6.2
|
||||||
|
dotenv==0.9.9
|
||||||
|
fastparquet==2024.11.0
|
||||||
|
fsspec==2025.9.0
|
||||||
|
idna==3.10
|
||||||
|
ijson==3.4.0
|
||||||
|
lxml==6.0.0
|
||||||
|
markdown-it-py==4.0.0
|
||||||
|
mdurl==0.1.2
|
||||||
|
numpy==2.3.2
|
||||||
|
packaging==25.0
|
||||||
|
pandas==2.3.1
|
||||||
|
pretty-tables==3.1.0
|
||||||
|
pyarrow==21.0.0
|
||||||
|
Pygments==2.19.2
|
||||||
|
python-dateutil==2.9.0.post0
|
||||||
python-dotenv==1.1.1
|
python-dotenv==1.1.1
|
||||||
Requests==2.32.5
|
pytz==2025.2
|
||||||
|
requests==2.32.4
|
||||||
|
six==1.17.0
|
||||||
|
tqdm==4.67.1
|
||||||
|
tzdata==2025.2
|
||||||
urllib3==2.5.0
|
urllib3==2.5.0
|
||||||
|
yarg==0.1.10
|
||||||
+44
-41
@@ -21,6 +21,8 @@ import ijson
|
|||||||
import os
|
import os
|
||||||
from bson import ObjectId
|
from bson import ObjectId
|
||||||
import datetime
|
import datetime
|
||||||
|
import tqdm
|
||||||
|
import sys
|
||||||
|
|
||||||
def pullPolicyExechistories(url, policiesnames, days, outputjson: bool):
|
def pullPolicyExechistories(url, policiesnames, days, outputjson: bool):
|
||||||
file_path = 'chunkinator.json'
|
file_path = 'chunkinator.json'
|
||||||
@@ -33,46 +35,47 @@ def pullPolicyExechistories(url, policiesnames, days, outputjson: bool):
|
|||||||
headers = {"X-APIKey": os.getenv('APIKEY')}
|
headers = {"X-APIKey": os.getenv('APIKEY')}
|
||||||
checkpoint = str(skipback(days))
|
checkpoint = str(skipback(days))
|
||||||
json_output = {'error': 'Success', 'response': {'exechistories': []}}
|
json_output = {'error': 'Success', 'response': {'exechistories': []}}
|
||||||
while True:
|
with tqdm.tqdm(file=sys.stdout, leave=True, total=10000, desc=f"Checkpoint Progess: {checkpoint}", colour="blue", initial=1) as filebar:
|
||||||
json_response_data = checkpoint_stomper(checkpoint, url, policiesnames, headers)
|
with tqdm.tqdm(file=sys.stdout, leave=True, total=100, desc=f"Total of {policiesnames} Complete: ") as pbar:
|
||||||
histories = json_response_data['response']['exechistories']
|
while True:
|
||||||
if not histories:
|
json_response_data = checkpoint_stomper(checkpoint, url, policiesnames, headers)
|
||||||
break
|
histories = json_response_data['response']['exechistories']
|
||||||
array_dividend = max(round(len(histories) / 20), 1)
|
filebar.total=len(histories)
|
||||||
match_found = False
|
if not histories:
|
||||||
for index, item in enumerate(histories[::array_dividend]):
|
break
|
||||||
if (datetime.date.today() - datetime.timedelta(days=days) <= datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()):
|
match_found = True
|
||||||
match_found = True
|
if match_found == True:
|
||||||
break
|
for index, item in enumerate(histories):
|
||||||
checkpoints_processed = round(len(histories) / array_dividend)
|
if index == len(histories) - 1:
|
||||||
if ( index + 1 ) < checkpoints_processed:
|
checkpoint = item['checkpoint']
|
||||||
print(ct.colorText(f"{index + 1}/{checkpoints_processed} checkpoint(s) from this execution have been processed with date match. Last Checkpoint: {checkpoint}", "blue"))
|
filebar.desc = f"Checkpoint Progress: {checkpoint}"
|
||||||
else:
|
break
|
||||||
print(ct.colorText(f"{index + 1}/{checkpoints_processed} checkpoint(s) Processed. Last Checkpoint: {checkpoint}", "blue"))
|
else:
|
||||||
if match_found == True:
|
if (datetime.date.today() - datetime.timedelta(days=days) > datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()):
|
||||||
for index, item in enumerate(histories):
|
pass
|
||||||
if index == len(histories) - 1:
|
else: json_output['response']['exechistories'].append(item)
|
||||||
print(ct.colorText(f"All Events Processed for {checkpoint}", "blue"))
|
filebar.update(1)
|
||||||
checkpoint = item['checkpoint']
|
filebar.refresh()
|
||||||
break
|
seen = {}
|
||||||
else:
|
if os.path.exists(file_path):
|
||||||
if (datetime.date.today() - datetime.timedelta(days=days) > datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()):
|
with open(file_path, 'r') as file:
|
||||||
pass
|
existing_data = json.load(file)
|
||||||
else: json_output['response']['exechistories'].append(item)
|
combined = existing_data['response']['exechistories'] + json_output['response']['exechistories']
|
||||||
seen = {}
|
else:
|
||||||
if os.path.exists(file_path):
|
combined = json_output['response']['exechistories']
|
||||||
with open(file_path, 'r') as file:
|
for item in combined:
|
||||||
existing_data = json.load(file)
|
key = (item.get('sha256'), item.get('filename'), item.get('hostname'))
|
||||||
combined = existing_data['response']['exechistories'] + json_output['response']['exechistories']
|
seen[key] = item
|
||||||
else:
|
deduplicated = list(seen.values())
|
||||||
combined = json_output['response']['exechistories']
|
with open(file_path, 'w') as file:
|
||||||
for item in combined:
|
json.dump({'error': 'Success', 'response': {'exechistories': deduplicated}}, file)
|
||||||
key = (item.get('sha256'), item.get('filename'), item.get('hostname'))
|
json_output['response']['exechistories'].clear()
|
||||||
seen[key] = item
|
date_diff = datetime.date.today() - datetime.datetime.strptime(item['datetime'].replace(' +0000 UTC', ''), '%Y-%m-%dT%H:%M:%SZ').date()
|
||||||
deduplicated = list(seen.values())
|
percentage_diff = (((days + 10) - date_diff.days) / (days + 10)) * 100
|
||||||
with open(file_path, 'w') as file:
|
pbar.n = round(percentage_diff)
|
||||||
json.dump({'error': 'Success', 'response': {'exechistories': deduplicated}}, file)
|
pbar.set_description_str(f"Total of {policiesnames} Complete: ")
|
||||||
json_output['response']['exechistories'].clear()
|
pbar.refresh()
|
||||||
|
filebar.n = 1
|
||||||
with open(file_path, 'r') as file:
|
with open(file_path, 'r') as file:
|
||||||
final_output = json.load(file)
|
final_output = json.load(file)
|
||||||
os.remove(file_path)
|
os.remove(file_path)
|
||||||
@@ -144,7 +147,7 @@ def skipback(days):
|
|||||||
Generate a MongoDB ObjectId for a given number of days ago from today.
|
Generate a MongoDB ObjectId for a given number of days ago from today.
|
||||||
Adds 1 extra day to the input to look further back.
|
Adds 1 extra day to the input to look further back.
|
||||||
"""
|
"""
|
||||||
adjusted_days = days + 1
|
adjusted_days = days + 10
|
||||||
date_days_ago = datetime.datetime.now(datetime.UTC) - datetime.timedelta(days=adjusted_days)
|
date_days_ago = datetime.datetime.now(datetime.UTC) - datetime.timedelta(days=adjusted_days)
|
||||||
timestamp = int(date_days_ago.timestamp())
|
timestamp = int(date_days_ago.timestamp())
|
||||||
hex_timestamp = format(timestamp, '08x')
|
hex_timestamp = format(timestamp, '08x')
|
||||||
|
|||||||
+187
-13
@@ -12,12 +12,15 @@
|
|||||||
#
|
#
|
||||||
# You should have received a copy of the GNU Affero General Public License
|
# You should have received a copy of the GNU Affero General Public License
|
||||||
# along with this program. If not, see <https://www.gnu.org/licenses/>.
|
# along with this program. If not, see <https://www.gnu.org/licenses/>.
|
||||||
|
import gc
|
||||||
|
import json
|
||||||
|
import os
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
import requests
|
import requests
|
||||||
import os
|
import utils.pathfunctions as pathf
|
||||||
import json
|
import utils.hashfunctions as hashf
|
||||||
import utils.pretty as ct
|
import utils.pretty as ct
|
||||||
|
from AirlockTools import tryToReadCSV
|
||||||
|
|
||||||
def aggregateHashes(executions_json) -> pd.DataFrame:
|
def aggregateHashes(executions_json) -> pd.DataFrame:
|
||||||
"""
|
"""
|
||||||
@@ -101,11 +104,9 @@ def augmentAggregatedHashes(url, agg_df: pd.DataFrame) -> pd.DataFrame:
|
|||||||
|
|
||||||
return aug_df
|
return aug_df
|
||||||
|
|
||||||
def categorizeHashes(aug_df: pd.DataFrame, threat_tolerance: int, untrusted_publishers: list):
|
def categorizeHashes(first_policy, second_policy, df: pd.DataFrame, threat_tolerance: int, untrusted_publishers, pups: list):
|
||||||
if untrusted_publishers is None:
|
if untrusted_publishers is None: untrusted_publishers = []
|
||||||
untrusted_publishers = []
|
if pups is None: pups = []
|
||||||
|
|
||||||
df = aug_df.copy()
|
|
||||||
|
|
||||||
def reputationtool(row):
|
def reputationtool(row):
|
||||||
val = row["reputation_scannermatch"]
|
val = row["reputation_scannermatch"]
|
||||||
@@ -126,14 +127,16 @@ def categorizeHashes(aug_df: pd.DataFrame, threat_tolerance: int, untrusted_publ
|
|||||||
mask_approved = (
|
mask_approved = (
|
||||||
(
|
(
|
||||||
(df["publisher"] != "Not Signed") &
|
(df["publisher"] != "Not Signed") &
|
||||||
~df["publisher"].isin(untrusted_publishers) &
|
~df["publisher"].str.contains(pathf.regulator(untrusted_publishers), case=False, na=False) &
|
||||||
~df["reputation_status"].isna()
|
~df["reputation_status"].isna() &
|
||||||
|
~df["description"].str.contains(pathf.regulator(pups), case=False, na=False)
|
||||||
) |
|
) |
|
||||||
(
|
(
|
||||||
(df["publisher"] == "Not Signed") &
|
(df["publisher"] == "Not Signed") &
|
||||||
~df["reputation_flag"] &
|
~df["reputation_flag"] &
|
||||||
~df["publisher"].isin(untrusted_publishers) &
|
~df["publisher"].str.contains(pathf.regulator(untrusted_publishers), case=False, na=False) &
|
||||||
~df["reputation_status"].isna()
|
~df["reputation_status"].isna() &
|
||||||
|
~df["description"].str.contains(pathf.regulator(pups), case=False, na=False)
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -141,7 +144,14 @@ def categorizeHashes(aug_df: pd.DataFrame, threat_tolerance: int, untrusted_publ
|
|||||||
approved_df = df[mask_approved]
|
approved_df = df[mask_approved]
|
||||||
unapproved_df = df[~(mask_needsreview | mask_approved)]
|
unapproved_df = df[~(mask_needsreview | mask_approved)]
|
||||||
|
|
||||||
return needsreview_df, approved_df, unapproved_df
|
needsreview_df.to_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet", index=False)
|
||||||
|
approved_df.to_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", index=False)
|
||||||
|
unapproved_df.to_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", index=False)
|
||||||
|
|
||||||
|
del needsreview_df
|
||||||
|
del approved_df
|
||||||
|
del unapproved_df
|
||||||
|
gc.collect()
|
||||||
|
|
||||||
def explode_and_deduplicate(df):
|
def explode_and_deduplicate(df):
|
||||||
df['sha256'] = df['sha256'].str.split(',')
|
df['sha256'] = df['sha256'].str.split(',')
|
||||||
@@ -199,3 +209,167 @@ def destinationHashes(
|
|||||||
# Concatenate results
|
# Concatenate results
|
||||||
df_hashdestination = pd.concat([df_paths, df_hashes], ignore_index=True)
|
df_hashdestination = pd.concat([df_paths, df_hashes], ignore_index=True)
|
||||||
return df_hashdestination
|
return df_hashdestination
|
||||||
|
|
||||||
|
def combineHashAndHist(path, first_policy, second_policy):
|
||||||
|
|
||||||
|
condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet")
|
||||||
|
df = pd.read_parquet(path)
|
||||||
|
|
||||||
|
#Pull hash info for the entries in the needs approval table
|
||||||
|
df = pd.merge(condensed_combo, df, on='sha256', how='inner')
|
||||||
|
|
||||||
|
#Rename Publisher, Keep and reorder columns we want
|
||||||
|
df = df.rename(columns={'publisher_x': 'publisher'})
|
||||||
|
df = df[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']]
|
||||||
|
df = df.sort_values(by='filename')
|
||||||
|
|
||||||
|
df.to_parquet(path, index=False)
|
||||||
|
del df
|
||||||
|
del condensed_combo
|
||||||
|
gc.collect()
|
||||||
|
|
||||||
|
def combineHashes(url, first_policy, second_policy):
|
||||||
|
combined_hashes = pd.DataFrame(columns=['sha256', 'publisher'])
|
||||||
|
hashes = []
|
||||||
|
try:
|
||||||
|
hash1 = pd.read_parquet(f"parquet\\execution_history_{first_policy}.parquet", columns=['sha256', 'publisher'])
|
||||||
|
pathf.inspect_parquet(f"parquet\\execution_history_{first_policy}.parquet")
|
||||||
|
if not hash1.empty:
|
||||||
|
hashes.append(hash1)
|
||||||
|
else:
|
||||||
|
print("⚠️ First dataframe is empty.")
|
||||||
|
except Exception as e:
|
||||||
|
print(f"❌ Error reading first Parquet file: {e}")
|
||||||
|
|
||||||
|
try:
|
||||||
|
hash2 = pd.read_parquet(f"parquet\\execution_history_{second_policy}.parquet", columns=['sha256', 'publisher'])
|
||||||
|
pathf.inspect_parquet(f"parquet\\execution_history_{second_policy}.parquet")
|
||||||
|
if not hash2.empty:
|
||||||
|
hashes.append(hash2)
|
||||||
|
else:
|
||||||
|
print("⚠️ Second dataframe is empty.")
|
||||||
|
except Exception as e:
|
||||||
|
print(f"❌ Error reading second Parquet file: {e}")
|
||||||
|
|
||||||
|
if hashes:
|
||||||
|
combined_hashes = pd.concat(hashes, ignore_index=True)
|
||||||
|
print(f"✅ Combined {len(combined_hashes)} hashes.")
|
||||||
|
else:
|
||||||
|
print("⚠️ No valid dataframes to combine.")
|
||||||
|
|
||||||
|
combined_hashes = combined_hashes.drop_duplicates(subset=['sha256'])
|
||||||
|
augmented_combo = hashf.augmentAggregatedHashes(url, combined_hashes)
|
||||||
|
|
||||||
|
numeric_reputation_cols = [
|
||||||
|
'reputation_scannermatch',
|
||||||
|
'reputation_scannercount',
|
||||||
|
'reputation_threatlevel'
|
||||||
|
]
|
||||||
|
|
||||||
|
for col in numeric_reputation_cols:
|
||||||
|
if col in augmented_combo.columns:
|
||||||
|
augmented_combo[col] = pd.to_numeric(augmented_combo[col].replace('N/A', pd.NA), errors='coerce')
|
||||||
|
|
||||||
|
augmented_combo = augmented_combo.rename(columns={'publisher_x': 'publisher'})
|
||||||
|
augmented_combo = augmented_combo[['sha256', 'publisher', 'description', 'productname', 'productversion',
|
||||||
|
'reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount',
|
||||||
|
'reputation_status', 'reputation_threatlevel', 'reputation_threatname',
|
||||||
|
'reputation_timestamp']]
|
||||||
|
augmented_combo = augmented_combo.sort_values(by=['publisher', 'description', 'productname'])
|
||||||
|
augmented_combo.to_parquet(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet", index=False)
|
||||||
|
|
||||||
|
del combined_hashes
|
||||||
|
del augmented_combo
|
||||||
|
gc.collect()
|
||||||
|
print(ct.colorText("Hash reputation info added to dataframe", "green"))
|
||||||
|
|
||||||
|
def condenseExecutions(first_policy,second_policy):
|
||||||
|
exe1 = pd.DataFrame()
|
||||||
|
exe2 = pd.DataFrame()
|
||||||
|
condensed_combo = pd.DataFrame()
|
||||||
|
|
||||||
|
try:
|
||||||
|
exe1 = pd.read_parquet(f"parquet\\execution_history_{first_policy}.parquet")
|
||||||
|
pathf.inspect_parquet(f"parquet\\execution_history_{first_policy}.parquet")
|
||||||
|
if not exe1.empty:
|
||||||
|
print()
|
||||||
|
else:
|
||||||
|
print("⚠️ First dataframe is empty.")
|
||||||
|
except Exception as e:
|
||||||
|
print(f"❌ Error reading first Parquet file: {e}")
|
||||||
|
|
||||||
|
try:
|
||||||
|
exe2 = pd.read_parquet(f"parquet\\execution_history_{second_policy}.parquet")
|
||||||
|
pathf.inspect_parquet(f"parquet\\execution_history_{second_policy}.parquet")
|
||||||
|
if not exe2.empty:
|
||||||
|
print()
|
||||||
|
else:
|
||||||
|
print("⚠️ Second dataframe is empty.")
|
||||||
|
except Exception as e:
|
||||||
|
print(f"❌ Error reading second Parquet file: {e}")
|
||||||
|
|
||||||
|
if not exe1.empty and not exe2.empty:
|
||||||
|
condensed_combo = pd.concat([exe1, exe2], ignore_index=True)
|
||||||
|
|
||||||
|
print(f"✅ Combined {len(condensed_combo)} hashes.")
|
||||||
|
elif exe1.empty:
|
||||||
|
condensed_combo = exe2
|
||||||
|
elif exe2.empty:
|
||||||
|
condensed_combo = exe1
|
||||||
|
else:
|
||||||
|
print("⚠️ No valid dataframes to combine.")
|
||||||
|
|
||||||
|
condensed_combo.to_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet", index=False)
|
||||||
|
del condensed_combo
|
||||||
|
gc.collect()
|
||||||
|
|
||||||
|
def divideSortedHashExecutions(first_policy,second_policy, pups):
|
||||||
|
|
||||||
|
combineHashAndHist(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet", first_policy, second_policy)
|
||||||
|
combineHashAndHist(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", first_policy, second_policy)
|
||||||
|
combineHashAndHist(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", first_policy, second_policy)
|
||||||
|
|
||||||
|
unknown = pd.read_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet")
|
||||||
|
good = pd.read_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet")
|
||||||
|
bad = pd.read_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet")
|
||||||
|
|
||||||
|
# Build regex pattern once
|
||||||
|
pattern = pathf.regulator(pups)
|
||||||
|
|
||||||
|
# Move matching rows from unknown and good to bad
|
||||||
|
bad = pd.concat([
|
||||||
|
bad,
|
||||||
|
unknown[unknown["filename"].str.contains(pattern, na=False)],
|
||||||
|
good[good["filename"].str.contains(pattern, na=False)]
|
||||||
|
], ignore_index=True)
|
||||||
|
|
||||||
|
# Remove matching rows from unknown and good
|
||||||
|
unknown = unknown[~unknown["filename"].str.contains(pattern, na=False)]
|
||||||
|
good = good[~good["filename"].str.contains(pattern, na=False)]
|
||||||
|
|
||||||
|
unknown.to_csv(f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv",index=False)
|
||||||
|
good.to_csv(f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv",index=False)
|
||||||
|
bad.to_csv(f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.csv",index=False)
|
||||||
|
|
||||||
|
ct.style_dataframe_dark(unknown, f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.html")
|
||||||
|
ct.style_dataframe_dark(good, f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.html")
|
||||||
|
ct.style_dataframe_dark(bad, f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.html")
|
||||||
|
|
||||||
|
def generatePreflights(first_policy, second_policy):
|
||||||
|
allhashes = pd.read_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet")
|
||||||
|
|
||||||
|
pathexclusions = tryToReadCSV(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv")
|
||||||
|
pathexclusions.to_parquet(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet", index=False)
|
||||||
|
|
||||||
|
allowbyhash = allhashes[~allhashes['sha256'].isin(pathexclusions['sha256'])]
|
||||||
|
|
||||||
|
allowbyhash.to_parquet(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet", index=False)
|
||||||
|
|
||||||
|
allowbyhash.sort_values(by=["filename"])
|
||||||
|
|
||||||
|
ct.style_dataframe_dark(allowbyhash, f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html")
|
||||||
|
ct.style_dataframe_dark(pathexclusions, f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html")
|
||||||
|
|
||||||
|
del allowbyhash
|
||||||
|
del pathexclusions
|
||||||
|
gc.collect()
|
||||||
+97
-65
@@ -12,76 +12,61 @@
|
|||||||
#
|
#
|
||||||
# You should have received a copy of the GNU Affero General Public License
|
# You should have received a copy of the GNU Affero General Public License
|
||||||
# along with this program. If not, see <https://www.gnu.org/licenses/>.
|
# along with this program. If not, see <https://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
import pandas as pd
|
|
||||||
import os
|
|
||||||
from itertools import chain
|
|
||||||
import ast
|
import ast
|
||||||
|
import gc
|
||||||
|
import os
|
||||||
|
import pandas as pd
|
||||||
import re
|
import re
|
||||||
|
import utils.pathfunctions as pathf
|
||||||
|
import utils.pretty as ct
|
||||||
|
from AirlockTools import tryToReadCSV
|
||||||
|
|
||||||
def split_path(path):
|
|
||||||
parts = []
|
|
||||||
while True:
|
|
||||||
head, tail = os.path.split(path)
|
|
||||||
if tail:
|
|
||||||
parts.insert(0, tail)
|
|
||||||
path = head
|
|
||||||
else:
|
|
||||||
if head:
|
|
||||||
parts.insert(0, head)
|
|
||||||
break
|
|
||||||
return parts
|
|
||||||
|
|
||||||
def local_common_pass(paths, min_parts=3):
|
def split_filepaths_grouped(df, col="filename", group_parts=4, min_parts=4):
|
||||||
results = {}
|
def clean_split(path):
|
||||||
paths_sorted = sorted(paths)
|
parts = os.path.normpath(path).split(os.sep)
|
||||||
for i, path in enumerate(paths_sorted):
|
# Remove leading empty strings caused by UNC paths
|
||||||
candidates = []
|
parts = [p for p in parts if p]
|
||||||
|
return parts
|
||||||
|
|
||||||
if i > 0:
|
df = df.copy()
|
||||||
try:
|
split_paths = df[col].apply(clean_split)
|
||||||
candidates.append(os.path.commonpath([path, paths_sorted[i-1]]))
|
|
||||||
except ValueError:
|
|
||||||
# different drives, skip
|
|
||||||
pass
|
|
||||||
if i < len(paths_sorted) - 1:
|
|
||||||
try:
|
|
||||||
candidates.append(os.path.commonpath([path, paths_sorted[i+1]]))
|
|
||||||
except ValueError:
|
|
||||||
# different drives, skip
|
|
||||||
pass
|
|
||||||
|
|
||||||
best = path
|
# Filter out paths with fewer than `min_parts` components
|
||||||
best_len = 0
|
df = df[split_paths.apply(lambda parts: len(parts) >= min_parts)].copy()
|
||||||
for c in candidates:
|
split_paths = split_paths[df.index] # Update split_paths to match filtered df
|
||||||
parts = split_path(c)
|
|
||||||
if len(parts) >= min_parts and len(parts) > best_len:
|
|
||||||
best = c
|
|
||||||
best_len = len(parts)
|
|
||||||
results[path] = best
|
|
||||||
return results
|
|
||||||
|
|
||||||
def add_longest_common_two_local(df, col="filename_x", new_col="longestcfp", min_parts=3):
|
df["group_key"] = split_paths.apply(lambda parts: os.sep.join(parts[:group_parts]))
|
||||||
dirs_series = df[col].astype(str).apply(os.path.dirname)
|
grouped = df.groupby("group_key")
|
||||||
first_pass = local_common_pass(dirs_series.tolist(), min_parts)
|
new_rows = []
|
||||||
second_pass = local_common_pass(list(first_pass.values()), min_parts)
|
|
||||||
df[new_col] = dirs_series.map(lambda d: second_pass[first_pass[d]])
|
|
||||||
return df
|
|
||||||
|
|
||||||
def export_groups_for_review(df, col, group_col, min_number_in_group, path_length_constant):
|
for _, group_df in grouped:
|
||||||
"""
|
paths = group_df[col].tolist()
|
||||||
Compute longest common paths, group filepaths, write CSV for review.
|
split_parts = [clean_split(p) for p in paths]
|
||||||
"""
|
|
||||||
df = df.drop_duplicates(subset=[col], keep='first')
|
|
||||||
df = add_longest_common_two_local(df, col=col, new_col=group_col)
|
|
||||||
grouped = df.groupby(group_col)[col].apply(list).reset_index()
|
|
||||||
grouped = grouped.sort_values(by=col)
|
|
||||||
print("Before filtering:", len(grouped))
|
|
||||||
grouped = grouped[grouped[col].apply(lambda x: len(x) >= min_number_in_group)]
|
|
||||||
filtered = grouped[grouped[group_col].apply(lambda x: len(os.path.normpath(x).split(os.sep)) >= path_length_constant)]
|
|
||||||
print("After filtering:", len(grouped))
|
|
||||||
|
|
||||||
return filtered, df
|
def longest_common_prefix(paths):
|
||||||
|
if not paths:
|
||||||
|
return []
|
||||||
|
prefix = paths[0]
|
||||||
|
for path in paths[1:]:
|
||||||
|
prefix = [a for a, b in zip(prefix, path) if a == b]
|
||||||
|
if not prefix:
|
||||||
|
break
|
||||||
|
return prefix
|
||||||
|
|
||||||
|
common_prefix = longest_common_prefix(split_parts)
|
||||||
|
prefix_str = os.sep.join(common_prefix)
|
||||||
|
|
||||||
|
for i, parts in enumerate(split_parts):
|
||||||
|
filename = parts[-1]
|
||||||
|
middle = os.sep.join(parts[len(common_prefix):-1]) if len(parts) > len(common_prefix) + 1 else ""
|
||||||
|
row = group_df.iloc[i].copy()
|
||||||
|
row["longestcfp"] = prefix_str
|
||||||
|
row["middle"] = middle
|
||||||
|
row["filename_only"] = filename
|
||||||
|
new_rows.append(row)
|
||||||
|
|
||||||
|
return pd.DataFrame(new_rows).drop(columns=["group_key"])
|
||||||
|
|
||||||
def mask_from_csv(df, csv_path, filepath_col):
|
def mask_from_csv(df, csv_path, filepath_col):
|
||||||
"""
|
"""
|
||||||
@@ -142,11 +127,58 @@ def inspect_parquet(path):
|
|||||||
|
|
||||||
def regulator(paths, case_insensitive=True):
|
def regulator(paths, case_insensitive=True):
|
||||||
"""
|
"""
|
||||||
Build a Python raw string regex that matches any of the given Windows path fragments.
|
Build a regex pattern that matches any of the given Windows path fragments.
|
||||||
"""
|
"""
|
||||||
escaped = [re.escape(p) for p in paths]
|
escaped = [re.escape(p) for p in paths]
|
||||||
pattern = "(?:" + "|".join(escaped) + ")"
|
pattern = "(?:" + "|".join(escaped) + ")"
|
||||||
if case_insensitive:
|
if case_insensitive:
|
||||||
pattern = pattern
|
pattern = "(?i)" + pattern # Add inline case-insensitive flag
|
||||||
print(f"Regulator is providing {pattern}")
|
print(f"Regulator is providing: {pattern}")
|
||||||
return f'r"{pattern}"'
|
return pattern
|
||||||
|
|
||||||
|
def generatePathReview(first_policy, second_policy, badpathparts, min_files_for_path):
|
||||||
|
|
||||||
|
if not os.path.exists(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet"):
|
||||||
|
|
||||||
|
df1 = tryToReadCSV(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv")
|
||||||
|
df2 = tryToReadCSV(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv")
|
||||||
|
|
||||||
|
all_approved_hashes = pd.concat([df1 , df2], ignore_index=True).sort_values(by=['filename'])
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
print(ct.colorText(f"Approved hash lists have been combined","green"))
|
||||||
|
|
||||||
|
all_approved_hashes.to_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet", index=False)
|
||||||
|
del all_approved_hashes
|
||||||
|
gc.collect()
|
||||||
|
|
||||||
|
if not os.path.exists(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet"):
|
||||||
|
all_approved_hashes = pd.read_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet")
|
||||||
|
print(ct.colorText(f"Beginning calculating longest common filepaths for path exceptions","green"))
|
||||||
|
|
||||||
|
haslcp = pathf.split_filepaths_grouped(all_approved_hashes)
|
||||||
|
haslcp.drop_duplicates()
|
||||||
|
|
||||||
|
forbidden = pathf.regulator(badpathparts, True)
|
||||||
|
forbidden_lcfp = haslcp["longestcfp"].str.contains(forbidden, na=False)
|
||||||
|
|
||||||
|
|
||||||
|
print(ct.colorText("Removing forbidden filepaths for path exceptions", "green"))
|
||||||
|
|
||||||
|
# Make a real DataFrame copy before modifying
|
||||||
|
lcp_not_forbidden = haslcp[~forbidden_lcfp].copy()
|
||||||
|
|
||||||
|
#For the review, drop down to only the columns we care, and then group by the commmon file path, consolidating and dropping dupes
|
||||||
|
lcp_not_forbidden_review = lcp_not_forbidden[['longestcfp', 'middle', 'filename_only', 'sha256']]
|
||||||
|
|
||||||
|
# Count unique sha256 per longestcfp
|
||||||
|
unique_sha_counts = lcp_not_forbidden_review.groupby('longestcfp')['sha256'].nunique().reset_index()
|
||||||
|
unique_sha_counts.columns = ['longestcfp', 'unique_sha256_count']
|
||||||
|
|
||||||
|
# Merge the count back into the original DataFrame
|
||||||
|
lcp_not_forbidden_review = lcp_not_forbidden_review.merge(unique_sha_counts, on='longestcfp', how='left')
|
||||||
|
lcp_not_forbidden_review = lcp_not_forbidden_review[lcp_not_forbidden_review['unique_sha256_count'] >= min_files_for_path]
|
||||||
|
|
||||||
|
lcp_not_forbidden_review.to_parquet(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet",index=False)
|
||||||
|
lcp_not_forbidden_review.to_csv(f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.csv",index=False)
|
||||||
+79
-20
@@ -12,19 +12,28 @@
|
|||||||
#
|
#
|
||||||
# You should have received a copy of the GNU Affero General Public License
|
# You should have received a copy of the GNU Affero General Public License
|
||||||
# along with this program. If not, see <https://www.gnu.org/licenses/>.
|
# along with this program. If not, see <https://www.gnu.org/licenses/>.
|
||||||
|
import gc
|
||||||
import requests
|
|
||||||
import json
|
import json
|
||||||
import os
|
import os
|
||||||
|
import pandas as pd
|
||||||
|
import re
|
||||||
|
import requests
|
||||||
import utils.pretty as ct
|
import utils.pretty as ct
|
||||||
|
import utils.allowlist
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
def addHash(policy, hash):
|
def addHash(policy, hash):
|
||||||
print(f"Adding the following hashes to {policy}:")
|
print(f"Adding the following hashes to {policy}:")
|
||||||
print(hash)
|
for p in hash:
|
||||||
|
print(p)
|
||||||
|
|
||||||
|
|
||||||
def addPath(policy, hash):
|
def addPath(policy, hash):
|
||||||
print(f"Adding the following Path Exclusions to {policy}:")
|
print(f"Adding the following Path Exclusions to {policy}:")
|
||||||
print(hash)
|
for p in hash:
|
||||||
|
print(p)
|
||||||
|
|
||||||
def addHashReal(url, allowlistID, hashlist):
|
def addHashReal(url, allowlistID, hashlist):
|
||||||
endpoint = url + '/v1/hash/application/add'
|
endpoint = url + '/v1/hash/application/add'
|
||||||
@@ -36,29 +45,79 @@ def addHashReal(url, allowlistID, hashlist):
|
|||||||
headers = {
|
headers = {
|
||||||
"X-APIKey": os.getenv('APIKEY')
|
"X-APIKey": os.getenv('APIKEY')
|
||||||
}
|
}
|
||||||
try:
|
payload = json.dumps(payload)
|
||||||
response = requests.request("POST", endpoint, headers=headers, data=payload, verify=False)
|
response = requests.request("POST", endpoint, headers=headers, data=payload, verify=False)
|
||||||
response.raise_for_status() # Raise an error for bad status codes
|
response.raise_for_status() # Raise an error for bad status codes
|
||||||
parse_text = json.loads(response.text)
|
parse_text = json.loads(response.text)
|
||||||
print(parse_text)
|
print(parse_text)
|
||||||
except requests.exceptions.RequestException as e:
|
|
||||||
return {"error": str(e)}
|
|
||||||
|
|
||||||
def addPathReal(url, grouplistID, pathlist):
|
def addPathReal(url, grouplistID, pathlist):
|
||||||
endpoint = url + '/v1/group/path/add'
|
endpoint = url + '/v1/group/path/add'
|
||||||
print(ct.colorText("[+] Grabbing All Categories", "cyan"))
|
print(ct.colorText("[+] Grabbing All Categories", "cyan"))
|
||||||
payload = {
|
payload = {
|
||||||
"applicationid" : grouplistID,
|
"groupid" : grouplistID,
|
||||||
"hashes" : pathlist
|
"path" : pathlist
|
||||||
}
|
}
|
||||||
headers = {
|
headers = {
|
||||||
"X-APIKey": os.getenv('APIKEY')
|
"X-APIKey": os.getenv('APIKEY')
|
||||||
}
|
}
|
||||||
try:
|
print(payload)
|
||||||
response = requests.request("POST", endpoint, headers=headers, data=payload, verify=False)
|
payload = json.dumps(payload)
|
||||||
response.raise_for_status() # Raise an error for bad status codes
|
response = requests.request("POST", endpoint, headers=headers, data=payload, verify=False)
|
||||||
parse_text = json.loads(response.text)
|
print(response.text)
|
||||||
print(parse_text)
|
|
||||||
except requests.exceptions.RequestException as e:
|
def getPolicyInfo(url, policy, days):
|
||||||
return {"error": str(e)}
|
executionhist_policy = pd.DataFrame()
|
||||||
|
exehist = utils.allowlist.pullPolicyExechistories(url, policy, days, True)
|
||||||
|
data = json.loads(exehist)
|
||||||
|
executionhist_policy = pd.DataFrame(data["response"]["exechistories"])
|
||||||
|
if not executionhist_policy.empty:
|
||||||
|
executionhist_policyxecutionhist_policy = executionhist_policy[['sha256', 'publisher', 'filename', 'hostname', 'username', 'pprocess', 'gprocess', 'commandline']]
|
||||||
|
executionhist_policy = executionhist_policy.drop_duplicates(subset=['sha256', 'filename', 'hostname'])
|
||||||
|
executionhist_policy = executionhist_policy.sort_values(by=['sha256', 'filename'])
|
||||||
|
executionhist_policy.to_parquet(f"parquet\\execution_history_{policy}.parquet", index=False)
|
||||||
|
print(ct.colorText(f"Staging of Execution history for policy: {policy} is complete", "green"))
|
||||||
|
del data
|
||||||
|
del exehist
|
||||||
|
gc.collect()
|
||||||
|
return executionhist_policy
|
||||||
|
|
||||||
|
def sendToPolicy(url, first_policy, second_policy, destination_name, destination_id, allowlist_parent_name, allowlist_parent_id, allowlist_child_name, allowlist_child_id):
|
||||||
|
pathexclusions = pd.read_parquet(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet")
|
||||||
|
allowbyhash = pd.read_parquet(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet")
|
||||||
|
|
||||||
|
ct.areYouSure()
|
||||||
|
confirmation = input(ct.colorText("Type 'I AGREE' to continue: ","white"))
|
||||||
|
|
||||||
|
if confirmation.strip().upper() == "I AGREE":
|
||||||
|
print(ct.colorText("Proceeding with the code...", "yellow"))
|
||||||
|
print(ct.colorText(f"Adding path exclusions to {destination_name}", "yellow"))
|
||||||
|
pathexcludelist = pathexclusions['longestcfp'].unique().tolist()
|
||||||
|
|
||||||
|
# Regex to match a Windows drive letter at the start (e.g., C:\)
|
||||||
|
drive_letter_pattern = re.compile(r'^[a-zA-Z]:\\')
|
||||||
|
|
||||||
|
# Processed list
|
||||||
|
processed_paths = [
|
||||||
|
(path if drive_letter_pattern.match(path) else f"\\\\{path}") + "**"
|
||||||
|
for path in pathexcludelist
|
||||||
|
]
|
||||||
|
addPath(url, destination_id,processed_paths)
|
||||||
|
|
||||||
|
print(ct.colorText(f"Adding hashes to {allowlist_parent_name}", "yellow"))
|
||||||
|
|
||||||
|
allowlist_parenthashlist = allowbyhash[allowbyhash['reputation_status'] == 'KNOWN']['sha256'].unique().tolist()
|
||||||
|
addHash(url, allowlist_parent_id,allowlist_parenthashlist)
|
||||||
|
|
||||||
|
print(ct.colorText(f"Adding hashes to {allowlist_child_name}", "yellow"))
|
||||||
|
allowlist_childhashlist = allowbyhash[allowbyhash['reputation_status'] == 'UNKNOWN']['sha256'].unique().tolist()
|
||||||
|
addHash(url, allowlist_child_id, allowlist_childhashlist)
|
||||||
|
|
||||||
|
ct.locked()
|
||||||
|
|
||||||
|
exit()
|
||||||
|
|
||||||
|
else:
|
||||||
|
print(ct.colorText("Operation aborted. You MUST EXPLICITLY AGREE to proceed.", "red"))
|
||||||
|
|
||||||
+126
@@ -1,3 +1,18 @@
|
|||||||
|
# Copyright (C) 2025 James Brotosky, Brandon Wickline
|
||||||
|
#
|
||||||
|
# This program is free software: you can redistribute it and/or modify
|
||||||
|
# it under the terms of the GNU Affero General Public License as published
|
||||||
|
# by the Free Software Foundation, either version 3 of the License, or
|
||||||
|
# (at your option) any later version.
|
||||||
|
#
|
||||||
|
# This program is distributed in the hope that it will be useful,
|
||||||
|
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
# GNU Affero General Public License for more details.
|
||||||
|
#
|
||||||
|
# You should have received a copy of the GNU Affero General Public License
|
||||||
|
# along with this program. If not, see <https://www.gnu.org/licenses/>.
|
||||||
|
import os
|
||||||
|
|
||||||
def colorText(text: str, color: str) -> str:
|
def colorText(text: str, color: str) -> str:
|
||||||
colors = {
|
colors = {
|
||||||
@@ -176,6 +191,117 @@ def displayIntro():
|
|||||||
print(colorText("======================== Welcome to the Airlock API Tool ========================", "cyan"))
|
print(colorText("======================== Welcome to the Airlock API Tool ========================", "cyan"))
|
||||||
print(colorText("=================================================================================", "cyan"))
|
print(colorText("=================================================================================", "cyan"))
|
||||||
|
|
||||||
|
def printEnforceChecklist(first_policy, second_policy, allowlist_child_name, allowlist_parent_name, destination_name):
|
||||||
|
|
||||||
|
print(colorText("\n --------------------------------------------------------------------", "cyan"))
|
||||||
|
print(colorText(" -------------------- Prepare to Enforce Policy ---------------------", "cyan"))
|
||||||
|
print(colorText(" --------------------------------------------------------------------", "cyan"))
|
||||||
|
print(colorText("\nSequentually follow these steps to prepare a policy for enforcement:", "white"))
|
||||||
|
|
||||||
|
print(colorText("\n1. Choose which originating policy or policies to move to enforcement", "cyan"))
|
||||||
|
if first_policy == " " and second_policy == " ":
|
||||||
|
print(colorText(f" [✗] No policies have been chosen","red"))
|
||||||
|
elif first_policy != " " and second_policy is first_policy:
|
||||||
|
print(colorText(f" [✓] {first_policy} has been selected,", "green"))
|
||||||
|
elif first_policy != " " and second_policy != " ":
|
||||||
|
print(colorText(f" [✓] {first_policy} has been selected as Policy 1","green"))
|
||||||
|
print(colorText(f" [✓] {second_policy} has been selected as Policy 2","green"))
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
print(colorText("2. Pull and stage event history, combine the histories, add hash info, then categorize the hashes", "cyan"))
|
||||||
|
|
||||||
|
if os.path.exists(f"parquet\\execution_history_{first_policy}.parquet"):
|
||||||
|
print(colorText(f" [✓] Execution history has been compiled for {first_policy}","green"))
|
||||||
|
elif not os.path.exists(f"parquet\\execution_history_{first_policy}.parquet"):
|
||||||
|
print(colorText(f" [✗] Execution history has not been compiled for {first_policy}","red"))
|
||||||
|
elif second_policy is not first_policy and os.path.exists(f"parquet\\execution_history_{second_policy}.parquet"):
|
||||||
|
print(colorText(f" [✓] Execution history has been compiled for {second_policy}","green"))
|
||||||
|
elif second_policy is not first_policy and not os.path.exists(f"parquet\\execution_history_{second_policy}.parquet"):
|
||||||
|
print(colorText(f" [✗] Execution history has not been compiled for {second_policy}","red"))
|
||||||
|
|
||||||
|
if os.path.exists(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"):
|
||||||
|
print(colorText(f" [✓] Hash Info has been added to the combined execution history", "green"))
|
||||||
|
else:
|
||||||
|
print(colorText(f" [✗] Hash Info has not been added to the combined execution history", "red"))
|
||||||
|
|
||||||
|
if os.path.exists(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") and os.path.exists(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") and os.path.exists(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet"):
|
||||||
|
print(colorText(f" [✓] Hashes have been cateogrized", "green"))
|
||||||
|
else:
|
||||||
|
print(colorText(f" [✗] Hashes have not been cateogrized", "red"))
|
||||||
|
|
||||||
|
if os.path.exists(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet"):
|
||||||
|
print(colorText(f" [✓] Execution history has been_combined_for {first_policy} and_{second_policy}", "green"))
|
||||||
|
else:
|
||||||
|
print(colorText(f" [✗] Execution history has not been_combined_for {first_policy} and_{second_policy}", "red"))
|
||||||
|
|
||||||
|
|
||||||
|
print(colorText(f"3. Manually review the files:","cyan"))
|
||||||
|
print(colorText(" 'needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv' and 'needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv'", "cyan"))
|
||||||
|
print(colorText(" Remove the rows containing hashes you do not approve of, and those you would not approve of without metarules.", "cyan"))
|
||||||
|
print(colorText(" If metarules need to be created, please make note of them, and remove the row from the csv.", "cyan"))
|
||||||
|
print(colorText(" When complete, save both csv files to the directory 'approved' and choose this option.","cyan"))
|
||||||
|
print(colorText(" This will combine these approved hashes with the automatically approved hashes and generate a list of paths to be reviewed", "cyan"))
|
||||||
|
|
||||||
|
if os.path.exists(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv") and os.path.exists(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv"):
|
||||||
|
print(colorText(" [✓] Reviewed hashes have been loaded","green"))
|
||||||
|
else:
|
||||||
|
print(colorText(" [✗] Reviewed hashes have not been loaded","red"))
|
||||||
|
|
||||||
|
if os.path.exists(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet"):
|
||||||
|
print(colorText(" [✓] The combined approved hashes list has been generated","green"))
|
||||||
|
else:
|
||||||
|
print(colorText(" [✗] The combined approved hashes list has not been generated","red"))
|
||||||
|
|
||||||
|
if os.path.exists(f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.csv"):
|
||||||
|
print(colorText(" [✓] Path review list created","green"))
|
||||||
|
else:
|
||||||
|
print(colorText(" [✗] Path review list has not been created","red"))
|
||||||
|
|
||||||
|
|
||||||
|
print(colorText(f"4. Manually review the file 'needs_approved\\paths_needing_review_{first_policy}_{second_policy}.csv'", "cyan"))
|
||||||
|
print(colorText(" Remove the rows containing path exclusions you do not approve of" , "cyan"))
|
||||||
|
print(colorText(" When complete, save the csv file to the directory 'approved'", "cyan"))
|
||||||
|
print(colorText(" Preflight Lists will be generated", "cyan"))
|
||||||
|
|
||||||
|
if os.path.exists(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv"):
|
||||||
|
print(colorText(" [✓] Reviewed path list detected","green"))
|
||||||
|
else:
|
||||||
|
print(colorText(" [✗] Path review list has not been detected","red"))
|
||||||
|
|
||||||
|
if os.path.exists(f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html"):
|
||||||
|
print(colorText(" [✓] Preflight Path Exclusion List has been generated","green"))
|
||||||
|
else:
|
||||||
|
print(colorText(" [✗] Preflight Path Exclusion List has not been generated","red"))
|
||||||
|
|
||||||
|
if os.path.exists(f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html"):
|
||||||
|
print(colorText(" [✓] Preflight hash approval list has been generated","green"))
|
||||||
|
else:
|
||||||
|
print(colorText(" [✗] Preflight hash approval list has not been generated","red"))
|
||||||
|
|
||||||
|
|
||||||
|
print(colorText(f"5. Choose the destination policy and parent and child allow list", "cyan"))
|
||||||
|
if allowlist_child_name == " " and allowlist_parent_name== " ":
|
||||||
|
print(colorText(f" [✗] No allowlists have been chosen","red"))
|
||||||
|
elif allowlist_parent_name != " " and allowlist_child_name != " " and allowlist_parent_name is allowlist_child_name:
|
||||||
|
print(colorText(f" [✓] [✗] Only {allowlist_parent_name} has been selected this is unusual, but potentially valid case, double check before proceeding,", "yellow"))
|
||||||
|
elif allowlist_parent_name != " " and allowlist_child_name != " " and allowlist_parent_name is not allowlist_child_name:
|
||||||
|
print(colorText(f" [✓] {allowlist_parent_name} has been selected as Parent Policy","green"))
|
||||||
|
print(colorText(f" [✓] {allowlist_child_name} has been selected as Child Policy","green"))
|
||||||
|
if destination_name == " ":
|
||||||
|
print(colorText(f" [✗] No destination policy has been chosen","red"))
|
||||||
|
else:
|
||||||
|
print(colorText(f" [✓] destination policy is {destination_name}","green"))
|
||||||
|
|
||||||
|
print(colorText(f"6. Liftoff ------------------------------------------------------", "cyan"))
|
||||||
|
print(colorText(f" Apply path exclusions according to allowed and approved paths", "cyan"))
|
||||||
|
print(colorText(f" Apply signed or attested hashes to Parent Allow List", "cyan"))
|
||||||
|
print(colorText(f" Apply approved, but unsigned hashes to the Child Allow List", "cyan"))
|
||||||
|
|
||||||
|
print(colorText("Q. Quit", "cyan"))
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
def areYouSure():
|
def areYouSure():
|
||||||
print(colorText(f"*******************************************************************************************************************************************","red"))
|
print(colorText(f"*******************************************************************************************************************************************","red"))
|
||||||
|
|||||||
Reference in New Issue
Block a user