# Copyright (C) 2025 James Brotosky, Brandon Wickline # # This program is free software: you can redistribute it and/or modify # it under the terms of the GNU Affero General Public License as published # by the Free Software Foundation, either version 3 of the License, or # (at your option) any later version. # # This program is distributed in the hope that it will be useful, # but WITHOUT ANY WARRANTY; without even the implied warranty of # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the # GNU Affero General Public License for more details. # # You should have received a copy of the GNU Affero General Public License # along with this program. If not, see . import pandas as pd import requests import os import json def aggregateHashes(executions_json) -> pd.DataFrame: """ Takes the executions, aggregates all the data with sha256 as primary, then returns aggregated dataframe """ data = json.loads(executions_json) df = pd.DataFrame(data["response"]["exechistories"]) if df.empty: return df print(df) # Aggregate by sha256, deduplicate lists, and preserve order agg_df = df.groupby("sha256").agg(lambda x: list(dict.fromkeys(x))).reset_index() # Add a column for the number of unique hostnames agg_df["num_devices"] = agg_df["hostname"].apply(len) # Sort by num_devices in descending order agg_df = agg_df.sort_values("num_devices", ascending=False) return agg_df def augmentAggregatedHashes(url, agg_df: pd.DataFrame) -> pd.DataFrame: """ Takes output of aggregatedHashes, queries API for those hashes, flattens response while keeping one row per hash, aggregate applications and baselines into lists, then merges results back into agg_df to create a """ endpoint = url + '/v1/hash/query' payload = { "hashes": agg_df['sha256'].tolist() } headers = {"X-APIKey": os.getenv('APIKEY')} payload = json.dumps(payload) response = requests.post(endpoint, headers=headers, data=payload, verify=False) data = response.json() results = data.get("response", {}).get("results", []) rows = [] for res in results: row = {"sha256": res.get("sha256"), "result": res.get("result")} if "data" in res: d = res["data"] for key in ["filename", "filepath", "description", "filesize", "md5", "productname", "productversion", "publisher", "createtime", "modtime", "sha128", "sha384", "sha512", "datetime"]: row[key] = d.get(key) row["applications"] = d.get("applications", []) row["baselines"] = d.get("baselines", []) reputation = d.get("reputation", {}) for k, v in reputation.items(): row[f"reputation_{k}"] = v rows.append(row) df_api = pd.DataFrame(rows) df = agg_df.merge(df_api, on="sha256", how="left") aug_df = df[['sha256', 'filename_x', 'description', 'productname', 'productversion', 'publisher_y', 'publisher_x', 'netdomain', 'hostname', 'username', 'pprocess', 'gprocess', 'commandline', 'reputation_lastseen', 'reputation_scannercount', 'reputation_scannermatch', 'reputation_status', 'reputation_threatlevel', 'reputation_threatname', 'reputation_timestamp']] return aug_df def categorizeHashes(aug_df: pd.DataFrame, threat_tolerance: int, untrusted_publishers: list): if untrusted_publishers is None: untrusted_publishers = [] df = aug_df.copy() def reputationtool(row): val = row["reputation_scannermatch"] if pd.isna(val) or val == "N/A": return row["publisher_y"] == "Not Signed" try: return int(val) > threat_tolerance except (ValueError, TypeError): return row["publisher_y"] == "Not Signed" df["reputation_flag"] = df.apply(reputationtool, axis=1) mask_needsreview = ( ((df["publisher_y"] == "Not Signed") & df["reputation_flag"]) | (df["reputation_status"] == "UNKNOWN") ) mask_approved = ( ( (df["publisher_y"] != "Not Signed") & ~df["publisher_y"].isin(untrusted_publishers) & ~df["reputation_status"].isna() ) | ( (df["publisher_y"] == "Not Signed") & ~df["reputation_flag"] & ~df["publisher_y"].isin(untrusted_publishers) & ~df["reputation_status"].isna() ) ) needsreview_df = df[mask_needsreview] approved_df = df[mask_approved] unapproved_df = df[~(mask_needsreview | mask_approved)] return needsreview_df, approved_df, unapproved_df