diff --git a/utils/hashfunctions.py b/utils/hashfunctions.py index 9f7bcca..6c41197 100644 --- a/utils/hashfunctions.py +++ b/utils/hashfunctions.py @@ -69,11 +69,12 @@ def augmentAggregatedHashes(url, agg_df: pd.DataFrame) -> pd.DataFrame: return aug_df def categorizeHashes(aug_df: pd.DataFrame, threat_tolerance: int, untrusted_publishers: list): - if untrusted_publishers is None: - untrusted_publishers = [] + if untrusted_publishers is None: + untrusted_publishers = [] - df = aug_df.copy() - def reputationtool(row, threat_tolerance): + df = aug_df.copy() + + def reputationtool(row, threat_tolerance): if row["reputation_scannermatch"] == "N/A": return True try: @@ -83,11 +84,15 @@ def categorizeHashes(aug_df: pd.DataFrame, threat_tolerance: int, untrusted_publ pass return False - mask_needsreview = (df["publisher_y"] == "Not Signed") & df.apply(lambda row: reputationtool(row, threat_tolerance), axis=1) - mask_approved = (df["publisher_y"] != "Not Signed") & (~df["publisher_y"].isin(untrusted_publishers)) + mask_needsreview = (df["publisher_y"] == "Not Signed") & df.apply(lambda row: reputationtool(row, threat_tolerance), axis=1) - needsreview_df = df[mask_needsreview] - approved_df = df[mask_approved] - remaining_df = df[~(mask_needsreview | mask_approved)] + mask_approved = ( + ((df["publisher_y"] != "Not Signed") & (~df["publisher_y"].isin(untrusted_publishers))) | + ((df["publisher_y"] == "Not Signed") & (~df.apply(lambda row: reputationtool(row, threat_tolerance), axis=1))) + ) - return needsreview_df, approved_df, remaining_df + needsreview_df = df[mask_needsreview] + approved_df = df[mask_approved] + remaining_df = df[~(mask_needsreview | mask_approved)] + + return needsreview_df, approved_df, remaining_df diff --git a/utils/pathfunctions.py b/utils/pathfunctions.py index 735c969..128da2a 100644 --- a/utils/pathfunctions.py +++ b/utils/pathfunctions.py @@ -41,18 +41,44 @@ def filepathInitialGroup(df: pd.DataFrame): return join_parts(prefix) # Step 6: Group directories by shared prefix using custom logic + """ + Loop through each directory path + directories: list of all directory paths. + groups: will hold lists of grouped directories. + used: tracks which directories have already been grouped. + """ directories = df["directory"].tolist() groups = [] used = set() + #For Each directory, compare it with others + """ + Skip if already grouped. + Start a new group with the current path. + parts_i is the list of folder names in the path (e.g., ["C:", "Users", "John", "Documents"]). + """ + for i, path in enumerate(directories): if path in used: continue group = [path] parts_i = get_parts(path) + + #Compare with all other directories: For each other directory, split it into parts and find the common prefix (shared folder structure). + """ + Logic: + If the directory is deep (>3 parts) and shares at least 3 parts → group it. + If it's exactly 3 parts long and shares at least 2 → group it. + Or, if it shares all but one part and is deep → group it. + These rules are designed to: + Group directories that are closely related in structure. + Avoid grouping unrelated paths that just happen to start similarly. + """ + for j in range(i + 1, len(directories)): parts_j = get_parts(directories[j]) common = os.path.commonprefix([parts_i, parts_j]) + #Apply grouping rules if (len(parts_i) > 3 and len(common) >= 3) or (len(parts_i) == 3 and len(common) >= 2): group.append(directories[j]) used.add(directories[j]) @@ -81,7 +107,7 @@ def filepathInitialGroup(df: pd.DataFrame): path_ineligible = grouped_df[grouped_df["depth"] <= 2].drop(columns=["depth"]) # Step 10: Move entries from eligible to ineligible if grouped_directory contains 'C:\Users' or 'c$\Users' - mask = path_eligible["grouped_directory"].str.contains(r"(?i)(?:\\Users|\\c\$\\Users)") + mask = path_eligible["grouped_directory"].str.contains(r"(?i)(?:\\Users|\\c\$\\Users|inetpub\\wwwroot|windows\\temp)", na=False) move_to_ineligible = path_eligible[mask] path_eligible = path_eligible[~mask] path_ineligible = pd.concat([path_ineligible, move_to_ineligible], ignore_index=True)