# Copyright (C) 2025 James Brotosky, Brandon Wickline # # This program is free software: you can redistribute it and/or modify # it under the terms of the GNU Affero General Public License as published # by the Free Software Foundation, either version 3 of the License, or # (at your option) any later version. # # This program is distributed in the hope that it will be useful, # but WITHOUT ANY WARRANTY; without even the implied warranty of # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the # GNU Affero General Public License for more details. # # You should have received a copy of the GNU Affero General Public License # along with this program. If not, see . import gc import json import os import pandas as pd import requests import utils.pathfunctions as pathf import utils.hashfunctions as hashf import utils.pretty as ct from AirlockTools import tryToReadCSV def aggregateHashes(executions_json) -> pd.DataFrame: """ Takes the executions, aggregates all the data with sha256 as primary, then returns aggregated dataframe """ data = json.loads(executions_json) df = pd.DataFrame(data["response"]["exechistories"]) if df.empty: return df print(df) # Aggregate by sha256, deduplicate lists, and preserve order agg_df = df.groupby("sha256").agg(lambda x: list(dict.fromkeys(x))).reset_index() # Add a column for the number of unique hostnames agg_df["num_devices"] = agg_df["hostname"].apply(len) # Sort by num_devices in descending order agg_df = agg_df.sort_values("num_devices", ascending=False) return agg_df def augmentAggregatedHashes(url, agg_df: pd.DataFrame) -> pd.DataFrame: """ Takes output of aggregatedHashes, queries API for those hashes, flattens response while keeping one row per hash, aggregate applications and baselines into lists, then merges results back into agg_df to create a """ if 'sha256' not in agg_df.columns or agg_df.empty: print("⚠️ 'sha256' column missing or DataFrame is empty. Skipping API query.") return agg_df.copy() # Return as-is to avoid breaking downstream logic endpoint = url + '/v1/hash/query' payload = { "hashes": agg_df['sha256'].tolist() } headers = {"X-APIKey": os.getenv('APIKEY')} payload = json.dumps(payload) response = requests.post(endpoint, headers=headers, data=payload, verify=False) data = response.json() results = data.get("response", {}).get("results", []) rows = [] for res in results: row = {"sha256": res.get("sha256"), "result": res.get("result")} if "data" in res: d = res["data"] for key in ["filename", "filepath", "description", "filesize", "md5", "productname", "productversion", "publisher", "createtime", "modtime", "sha128", "sha384", "sha512", "datetime"]: row[key] = d.get(key) row["applications"] = d.get("applications", []) row["baselines"] = d.get("baselines", []) reputation = d.get("reputation", {}) for k, v in reputation.items(): row[f"reputation_{k}"] = v rows.append(row) df_api = pd.DataFrame(rows) if 'sha256' not in df_api.columns: print("⚠️ API response missing 'sha256'. Skipping merge.") return agg_df.copy() df = agg_df.merge(df_api, on="sha256", how="left") # Only include columns that exist to avoid KeyErrors expected_columns = ['sha256', 'filename_x', 'description', 'productname', 'productversion', 'publisher_y', 'publisher_x', 'netdomain', 'hostname', 'username', 'reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount', 'reputation_status', 'reputation_threatlevel', 'reputation_threatname', 'reputation_timestamp', 'pprocess', 'gprocess', 'commandline'] available_columns = [col for col in expected_columns if col in df.columns] aug_df = df[available_columns] return aug_df def categorizeHashes(first_policy, second_policy, df: pd.DataFrame, threat_tolerance: int, untrusted_publishers, pups: list): if untrusted_publishers is None: untrusted_publishers = [] if pups is None: pups = [] def reputationtool(row): val = row["reputation_scannermatch"] if pd.isna(val) or val == "N/A": return row["publisher"] == "Not Signed" try: return int(val) > threat_tolerance except (ValueError, TypeError): return row["publisher"] == "Not Signed" df["reputation_flag"] = df.apply(reputationtool, axis=1) mask_needsreview = ( ((df["publisher"] == "Not Signed") & df["reputation_flag"]) | (df["reputation_status"] == "UNKNOWN") ) mask_approved = ( ( (df["publisher"] != "Not Signed") & ~df["publisher"].str.contains(pathf.regulator(untrusted_publishers), case=False, na=False) & ~df["reputation_status"].isna() & ~df["description"].str.contains(pathf.regulator(pups), case=False, na=False) ) | ( (df["publisher"] == "Not Signed") & ~df["reputation_flag"] & ~df["publisher"].str.contains(pathf.regulator(untrusted_publishers), case=False, na=False) & ~df["reputation_status"].isna() & ~df["description"].str.contains(pathf.regulator(pups), case=False, na=False) ) ) needsreview_df = df[mask_needsreview] approved_df = df[mask_approved] unapproved_df = df[~(mask_needsreview | mask_approved)] needsreview_df.to_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet", index=False) approved_df.to_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", index=False) unapproved_df.to_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", index=False) del needsreview_df del approved_df del unapproved_df gc.collect() def explode_and_deduplicate(df): df['sha256'] = df['sha256'].str.split(',') df = df.explode('sha256') return df.drop_duplicates().reset_index(drop=True) def clean_sha256(df, column='sha256'): """Discard quotes, brackets, and whitespace from sha256 values.""" df[column] = df[column].astype(str).str.strip("'[]\" ") return df def destinationHashes( df_approved_paths: pd.DataFrame, df_approved_hashes: pd.DataFrame, df_hashes_auto_approved: pd.DataFrame, df_hashes_manually_approved: pd.DataFrame, ): # Deduplicate and explode all input DataFrames df_approved_paths = explode_and_deduplicate(df_approved_paths) df_approved_hashes = explode_and_deduplicate(df_approved_hashes) df_hashes_auto_approved = explode_and_deduplicate(df_hashes_auto_approved) df_hashes_manually_approved = explode_and_deduplicate(df_hashes_manually_approved) # Clean sha256 values in all relevant DataFrames df_approved_hashes = clean_sha256(df_approved_hashes) df_hashes_auto_approved = clean_sha256(df_hashes_auto_approved) df_hashes_manually_approved = clean_sha256(df_hashes_manually_approved) # Create sets for faster lookup auto_approved_sha256 = set(df_hashes_auto_approved['sha256'].values) manually_approved_sha256 = set(df_hashes_manually_approved['sha256'].values) # Debug: Print unmatched hashes unmatched = set(df_approved_hashes['sha256']) - (auto_approved_sha256 | manually_approved_sha256) print(f"Unmatched hashes: {unmatched}") # Process df_approved_paths df_paths = df_approved_paths.assign(destination='Path Exclusion') df_paths = df_paths[['sha256', 'description', 'destination', 'grouped_directory', 'filename']] # Process df_approved_hashes df_hashes = df_approved_hashes.copy() df_hashes['destination'] = df_hashes['sha256'].apply( lambda x: 'Parent Policy Baseline' if x in auto_approved_sha256 else ('Child Policy Allowlist' if x in manually_approved_sha256 else None) ) df_hashes = df_hashes.dropna(subset=['destination']) df_hashes = df_hashes.assign(grouped_directory=None) # Use 'filename_x' only if it exists, otherwise fallback to 'filename' filename_col = 'filename_x' if 'filename_x' in df_hashes.columns else 'filename' selected_cols = ['sha256', 'description', 'destination', 'grouped_directory', filename_col] df_hashes = df_hashes[selected_cols] # Concatenate results df_hashdestination = pd.concat([df_paths, df_hashes], ignore_index=True) return df_hashdestination def combineHashAndHist(path, first_policy, second_policy): condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet") df = pd.read_parquet(path) #Pull hash info for the entries in the needs approval table df = pd.merge(condensed_combo, df, on='sha256', how='inner') #Rename Publisher, Keep and reorder columns we want df = df.rename(columns={'publisher_x': 'publisher'}) df = df[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']] df = df.sort_values(by='filename') df.to_parquet(path, index=False) del df del condensed_combo gc.collect() def combineHashes(url, first_policy, second_policy): combined_hashes = pd.DataFrame(columns=['sha256', 'publisher']) hashes = [] try: hash1 = pd.read_parquet(f"parquet\\execution_history_{first_policy}.parquet", columns=['sha256', 'publisher']) pathf.inspect_parquet(f"parquet\\execution_history_{first_policy}.parquet") if not hash1.empty: hashes.append(hash1) else: print("⚠️ First dataframe is empty.") except Exception as e: print(f"❌ Error reading first Parquet file: {e}") try: hash2 = pd.read_parquet(f"parquet\\execution_history_{second_policy}.parquet", columns=['sha256', 'publisher']) pathf.inspect_parquet(f"parquet\\execution_history_{second_policy}.parquet") if not hash2.empty: hashes.append(hash2) else: print("⚠️ Second dataframe is empty.") except Exception as e: print(f"❌ Error reading second Parquet file: {e}") if hashes: combined_hashes = pd.concat(hashes, ignore_index=True) print(f"✅ Combined {len(combined_hashes)} hashes.") else: print("⚠️ No valid dataframes to combine.") combined_hashes = combined_hashes.drop_duplicates(subset=['sha256']) augmented_combo = hashf.augmentAggregatedHashes(url, combined_hashes) numeric_reputation_cols = [ 'reputation_scannermatch', 'reputation_scannercount', 'reputation_threatlevel' ] for col in numeric_reputation_cols: if col in augmented_combo.columns: augmented_combo[col] = pd.to_numeric(augmented_combo[col].replace('N/A', pd.NA), errors='coerce') augmented_combo = augmented_combo.rename(columns={'publisher_x': 'publisher'}) augmented_combo = augmented_combo[['sha256', 'publisher', 'description', 'productname', 'productversion', 'reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount', 'reputation_status', 'reputation_threatlevel', 'reputation_threatname', 'reputation_timestamp']] augmented_combo = augmented_combo.sort_values(by=['publisher', 'description', 'productname']) augmented_combo.to_parquet(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet", index=False) del combined_hashes del augmented_combo gc.collect() print(ct.colorText("Hash reputation info added to dataframe", "green")) def condenseExecutions(first_policy,second_policy): exe1 = pd.DataFrame() exe2 = pd.DataFrame() condensed_combo = pd.DataFrame() try: exe1 = pd.read_parquet(f"parquet\\execution_history_{first_policy}.parquet") pathf.inspect_parquet(f"parquet\\execution_history_{first_policy}.parquet") if not exe1.empty: print() else: print("⚠️ First dataframe is empty.") except Exception as e: print(f"❌ Error reading first Parquet file: {e}") try: exe2 = pd.read_parquet(f"parquet\\execution_history_{second_policy}.parquet") pathf.inspect_parquet(f"parquet\\execution_history_{second_policy}.parquet") if not exe2.empty: print() else: print("⚠️ Second dataframe is empty.") except Exception as e: print(f"❌ Error reading second Parquet file: {e}") if not exe1.empty and not exe2.empty: condensed_combo = pd.concat([exe1, exe2], ignore_index=True) print(f"✅ Combined {len(condensed_combo)} hashes.") elif exe1.empty: condensed_combo = exe2 elif exe2.empty: condensed_combo = exe1 else: print("⚠️ No valid dataframes to combine.") condensed_combo.to_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet", index=False) del condensed_combo gc.collect() def divideSortedHashExecutions(first_policy,second_policy, pups): combineHashAndHist(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet", first_policy, second_policy) combineHashAndHist(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", first_policy, second_policy) combineHashAndHist(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", first_policy, second_policy) unknown = pd.read_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") good = pd.read_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") bad = pd.read_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet") # Build regex pattern once pattern = pathf.regulator(pups) # Move matching rows from unknown and good to bad bad = pd.concat([ bad, unknown[unknown["filename"].str.contains(pattern, na=False)], good[good["filename"].str.contains(pattern, na=False)] ], ignore_index=True) # Remove matching rows from unknown and good unknown = unknown[~unknown["filename"].str.contains(pattern, na=False)] good = good[~good["filename"].str.contains(pattern, na=False)] unknown.to_csv(f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv",index=False) good.to_csv(f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv",index=False) bad.to_csv(f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.csv",index=False) ct.style_dataframe_dark(unknown, f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.html") ct.style_dataframe_dark(good, f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.html") ct.style_dataframe_dark(bad, f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.html") def generatePreflights(first_policy, second_policy): allhashes = pd.read_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet") pathexclusions = tryToReadCSV(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv") pathexclusions.to_parquet(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet", index=False) allowbyhash = allhashes[~allhashes['sha256'].isin(pathexclusions['sha256'])] allowbyhash.to_parquet(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet", index=False) allowbyhash.sort_values(by=["filename"]) ct.style_dataframe_dark(allowbyhash, f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html") ct.style_dataframe_dark(pathexclusions, f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html") del allowbyhash del pathexclusions gc.collect()