Files
AirlockTools/utils/hashfunctions.py
T

404 lines
17 KiB
Python

# Copyright (C) 2025 James Brotosky, Brandon Wickline
#
# This program is free software: you can redistribute it and/or modify
# it under the terms of the GNU Affero General Public License as published
# by the Free Software Foundation, either version 3 of the License, or
# (at your option) any later version.
#
# This program is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# GNU Affero General Public License for more details.
#
# You should have received a copy of the GNU Affero General Public License
# along with this program. If not, see <https://www.gnu.org/licenses/>.
#Local Imports
import utils.hashfunctions as hashf
import utils.pathfunctions as pathf
import utils.pretty as ct
from AirlockTools import tryToReadCSV
#Standard Libary Imports:
import gc
import json
import os
#3rd Party Imports:
import pandas as pd
import requests
def aggregateHashes(executions_json) -> pd.DataFrame:
"""
Takes the executions, aggregates all the data with sha256 as primary, then returns aggregated dataframe
"""
data = json.loads(executions_json)
df = pd.DataFrame(data["response"]["exechistories"])
if df.empty:
return df
print(df)
# Aggregate by sha256, deduplicate lists, and preserve order
agg_df = df.groupby("sha256").agg(lambda x: list(dict.fromkeys(x))).reset_index()
# Add a column for the number of unique hostnames
agg_df["num_devices"] = agg_df["hostname"].apply(len)
# Sort by num_devices in descending order
agg_df = agg_df.sort_values("num_devices", ascending=False)
return agg_df
def augmentAggregatedHashes(url, agg_df: pd.DataFrame) -> pd.DataFrame:
"""
Takes output of aggregatedHashes, queries API for those hashes, flattens response while keeping one row per hash,
aggregate applications and baselines into lists, then merges results back into agg_df to create a
"""
if 'sha256' not in agg_df.columns or agg_df.empty:
print("⚠️ 'sha256' column missing or DataFrame is empty. Skipping API query.")
return agg_df.copy() # Return as-is to avoid breaking downstream logic
endpoint = url + '/v1/hash/query'
payload = {
"hashes": agg_df['sha256'].tolist()
}
headers = {"X-APIKey": os.getenv('APIKEY')}
payload = json.dumps(payload)
response = requests.post(endpoint, headers=headers, data=payload, verify=False)
data = response.json()
results = data.get("response", {}).get("results", [])
rows = []
for res in results:
row = {"sha256": res.get("sha256"), "result": res.get("result")}
if "data" in res:
d = res["data"]
for key in ["filename", "filepath", "description", "filesize", "md5",
"productname", "productversion", "publisher", "createtime", "modtime",
"sha128", "sha384", "sha512", "datetime"]:
row[key] = d.get(key)
row["applications"] = d.get("applications", [])
row["baselines"] = d.get("baselines", [])
reputation = d.get("reputation", {})
for k, v in reputation.items():
row[f"reputation_{k}"] = v
rows.append(row)
df_api = pd.DataFrame(rows)
if 'sha256' not in df_api.columns:
print("⚠️ API response missing 'sha256'. Skipping merge.")
return agg_df.copy()
df = agg_df.merge(df_api, on="sha256", how="left")
# Only include columns that exist to avoid KeyErrors
expected_columns = ['sha256', 'filename_x', 'description', 'productname', 'productversion',
'publisher_y', 'publisher_x', 'netdomain', 'hostname', 'username',
'reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount',
'reputation_status', 'reputation_threatlevel', 'reputation_threatname',
'reputation_timestamp', 'pprocess', 'gprocess', 'commandline']
available_columns = [col for col in expected_columns if col in df.columns]
aug_df = df[available_columns]
return aug_df
def categorizeHashes(first_policy, second_policy, df: pd.DataFrame, threat_tolerance: int, untrusted_publishers, pups: list):
if untrusted_publishers is None: untrusted_publishers = []
if pups is None: pups = []
def reputationtool(row):
val = row["reputation_scannermatch"]
if pd.isna(val) or val == "N/A":
return row["publisher"] == "Not Signed"
try:
return int(val) > threat_tolerance
except (ValueError, TypeError):
return row["publisher"] == "Not Signed"
df["reputation_flag"] = df.apply(reputationtool, axis=1)
mask_needsreview = (
((df["publisher"] == "Not Signed") & df["reputation_flag"]) |
(df["reputation_status"] == "UNKNOWN")
)
mask_approved = (
(
(df["publisher"] != "Not Signed") &
~df["publisher"].str.contains(pathf.regulator(untrusted_publishers), case=False, na=False) &
~df["reputation_status"].isna() &
~df["description"].str.contains(pathf.regulator(pups), case=False, na=False)
) |
(
(df["publisher"] == "Not Signed") &
~df["reputation_flag"] &
~df["publisher"].str.contains(pathf.regulator(untrusted_publishers), case=False, na=False) &
~df["reputation_status"].isna() &
~df["description"].str.contains(pathf.regulator(pups), case=False, na=False)
)
)
needsreview_df = df[mask_needsreview]
approved_df = df[mask_approved]
unapproved_df = df[~(mask_needsreview | mask_approved)]
needsreview_df.to_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet", index=False)
approved_df.to_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", index=False)
unapproved_df.to_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", index=False)
del needsreview_df
del approved_df
del unapproved_df
gc.collect()
def explode_and_deduplicate(df):
df['sha256'] = df['sha256'].str.split(',')
df = df.explode('sha256')
return df.drop_duplicates().reset_index(drop=True)
def clean_sha256(df, column='sha256'):
"""Discard quotes, brackets, and whitespace from sha256 values."""
df[column] = df[column].astype(str).str.strip("'[]\" ")
return df
def destinationHashes(
df_approved_paths: pd.DataFrame,
df_approved_hashes: pd.DataFrame,
df_hashes_auto_approved: pd.DataFrame,
df_hashes_manually_approved: pd.DataFrame,
):
# Deduplicate and explode all input DataFrames
df_approved_paths = explode_and_deduplicate(df_approved_paths)
df_approved_hashes = explode_and_deduplicate(df_approved_hashes)
df_hashes_auto_approved = explode_and_deduplicate(df_hashes_auto_approved)
df_hashes_manually_approved = explode_and_deduplicate(df_hashes_manually_approved)
# Clean sha256 values in all relevant DataFrames
df_approved_hashes = clean_sha256(df_approved_hashes)
df_hashes_auto_approved = clean_sha256(df_hashes_auto_approved)
df_hashes_manually_approved = clean_sha256(df_hashes_manually_approved)
# Create sets for faster lookup
auto_approved_sha256 = set(df_hashes_auto_approved['sha256'].values)
manually_approved_sha256 = set(df_hashes_manually_approved['sha256'].values)
# Debug: Print unmatched hashes
unmatched = set(df_approved_hashes['sha256']) - (auto_approved_sha256 | manually_approved_sha256)
print(f"Unmatched hashes: {unmatched}")
# Process df_approved_paths
df_paths = df_approved_paths.assign(destination='Path Exclusion')
df_paths = df_paths[['sha256', 'description', 'destination', 'grouped_directory', 'filename']]
# Process df_approved_hashes
df_hashes = df_approved_hashes.copy()
df_hashes['destination'] = df_hashes['sha256'].apply(
lambda x: 'Parent Policy Baseline' if x in auto_approved_sha256
else ('Child Policy Allowlist' if x in manually_approved_sha256 else None)
)
df_hashes = df_hashes.dropna(subset=['destination'])
df_hashes = df_hashes.assign(grouped_directory=None)
# Use 'filename_x' only if it exists, otherwise fallback to 'filename'
filename_col = 'filename_x' if 'filename_x' in df_hashes.columns else 'filename'
selected_cols = ['sha256', 'description', 'destination', 'grouped_directory', filename_col]
df_hashes = df_hashes[selected_cols]
# Concatenate results
df_hashdestination = pd.concat([df_paths, df_hashes], ignore_index=True)
return df_hashdestination
def combineHashAndHist(path, first_policy, second_policy):
condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet")
df = pd.read_parquet(path)
#Pull hash info for the entries in the needs approval table
df = pd.merge(condensed_combo, df, on='sha256', how='inner')
#Rename Publisher, Keep and reorder columns we want
df = df.rename(columns={'publisher_x': 'publisher'})
df = df[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']]
df = df.sort_values(by='filename')
df.to_parquet(path, index=False)
del df
del condensed_combo
gc.collect()
def combineHashes(url, first_policy, second_policy):
combined_hashes = pd.DataFrame(columns=['sha256', 'publisher'])
hashes = []
try:
hash1 = pd.read_parquet(f"parquet\\execution_history_{first_policy}.parquet", columns=['sha256', 'publisher'])
pathf.inspect_parquet(f"parquet\\execution_history_{first_policy}.parquet")
if not hash1.empty:
hashes.append(hash1)
else:
print("⚠️ First dataframe is empty.")
except Exception as e:
print(f"❌ Error reading first Parquet file: {e}")
try:
hash2 = pd.read_parquet(f"parquet\\execution_history_{second_policy}.parquet", columns=['sha256', 'publisher'])
pathf.inspect_parquet(f"parquet\\execution_history_{second_policy}.parquet")
if not hash2.empty:
hashes.append(hash2)
else:
print("⚠️ Second dataframe is empty.")
except Exception as e:
print(f"❌ Error reading second Parquet file: {e}")
if hashes:
combined_hashes = pd.concat(hashes, ignore_index=True)
print(f"✅ Combined {len(combined_hashes)} hashes.")
else:
print("⚠️ No valid dataframes to combine.")
combined_hashes = combined_hashes.drop_duplicates(subset=['sha256'])
augmented_combo = hashf.augmentAggregatedHashes(url, combined_hashes)
numeric_reputation_cols = [
'reputation_scannermatch',
'reputation_scannercount',
'reputation_threatlevel'
]
for col in numeric_reputation_cols:
if col in augmented_combo.columns:
augmented_combo[col] = pd.to_numeric(augmented_combo[col].replace('N/A', pd.NA), errors='coerce')
augmented_combo = augmented_combo.rename(columns={'publisher_x': 'publisher'})
augmented_combo = augmented_combo[['sha256', 'publisher', 'description', 'productname', 'productversion',
'reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount',
'reputation_status', 'reputation_threatlevel', 'reputation_threatname',
'reputation_timestamp']]
augmented_combo = augmented_combo.sort_values(by=['publisher', 'description', 'productname'])
augmented_combo.to_parquet(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet", index=False)
del combined_hashes
del augmented_combo
gc.collect()
print(ct.colorText("Hash reputation info added to dataframe", "green"))
def condenseExecutions(first_policy,second_policy):
exe1 = pd.DataFrame()
exe2 = pd.DataFrame()
condensed_combo = pd.DataFrame()
try:
exe1 = pd.read_parquet(f"parquet\\execution_history_{first_policy}.parquet")
pathf.inspect_parquet(f"parquet\\execution_history_{first_policy}.parquet")
if not exe1.empty:
print()
else:
print("⚠️ First dataframe is empty.")
except Exception as e:
print(f"❌ Error reading first Parquet file: {e}")
try:
exe2 = pd.read_parquet(f"parquet\\execution_history_{second_policy}.parquet")
pathf.inspect_parquet(f"parquet\\execution_history_{second_policy}.parquet")
if not exe2.empty:
print()
else:
print("⚠️ Second dataframe is empty.")
except Exception as e:
print(f"❌ Error reading second Parquet file: {e}")
if not exe1.empty and not exe2.empty:
condensed_combo = pd.concat([exe1, exe2], ignore_index=True)
print(f"✅ Combined {len(condensed_combo)} hashes.")
elif exe1.empty:
condensed_combo = exe2
elif exe2.empty:
condensed_combo = exe1
else:
print("⚠️ No valid dataframes to combine.")
condensed_combo.to_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet", index=False)
del condensed_combo
gc.collect()
def divideSortedHashExecutions(first_policy,second_policy, pups):
combineHashAndHist(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet", first_policy, second_policy)
combineHashAndHist(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", first_policy, second_policy)
combineHashAndHist(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", first_policy, second_policy)
unknown = pd.read_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet")
good = pd.read_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet")
bad = pd.read_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet")
# Build regex pattern once
pattern = pathf.regulator(pups)
# Move matching rows from unknown and good to bad
bad = pd.concat([
bad,
unknown[unknown["filename"].str.contains(pattern, na=False)],
good[good["filename"].str.contains(pattern, na=False)]
], ignore_index=True)
# Remove matching rows from unknown and good
unknown = unknown[~unknown["filename"].str.contains(pattern, na=False)]
good = good[~good["filename"].str.contains(pattern, na=False)]
unknown.to_csv(f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv",index=False)
good.to_csv(f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv",index=False)
bad.to_csv(f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.csv",index=False)
ct.style_dataframe_dark(unknown, f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.html")
ct.style_dataframe_dark(good, f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.html")
ct.style_dataframe_dark(bad, f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.html")
def generatePreflights(first_policy, second_policy):
allhashes = pd.read_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet")
primarypathexclusions = tryToReadCSV(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv")
secondarypathexclusions = tryToReadCSV(f"approved\\secondary_paths_{first_policy}_{second_policy}.csv")
pathexclusions = pd.concat([primarypathexclusions, secondarypathexclusions], ignore_index=True)
publishers = tryToReadCSV(f"approved\\publishers_{first_policy}_{second_policy}.csv")
publishers.to_parquet(f"parquet\\publishers_{first_policy}_{second_policy}.parquet", index=False)
allowbyhash = allhashes[~allhashes['sha256'].isin(pathexclusions['sha256'])]
allowbyhash.to_parquet(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet", index=False)
allowbyhash.sort_values(by=["filename"])
ct.style_dataframe_dark(allowbyhash, f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html")
ct.style_dataframe_dark(pathexclusions, f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html")
ct.style_dataframe_dark(publishers, f"preflight\\publishers_{first_policy}_{second_policy}.html")
del allowbyhash
del pathexclusions
gc.collect()
def generatePublist(first_policy,second_policy,bad_publisher_list):
try:
publist = pd.read_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet", columns=['publisher'])
except Exception as e:
print(f"Error reading parquet file: {e}")
publist = pd.DataFrame()
#Drop all not signed, only keep unique values
publist = publist[publist['publisher'] != "Not Signed"].drop_duplicates(subset='publisher')
#Remove Bad publisher if somehow they made it this far
pattern = pathf.regulator(bad_publisher_list)
publist = publist[~publist["publisher"].str.contains(pattern, na=False)]
publist.to_csv(f"needs_approved\\publishers_{first_policy}_{second_policy}.csv", index=False)