Files
AirlockTools/utils/pathfunctions.py
T
2025-08-26 17:12:06 -04:00

134 lines
5.4 KiB
Python

# Copyright (C) 2025 James Brotosky, Brandon Wickline
#
# This program is free software: you can redistribute it and/or modify
# it under the terms of the GNU Affero General Public License as published
# by the Free Software Foundation, either version 3 of the License, or
# (at your option) any later version.
#
# This program is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# GNU Affero General Public License for more details.
#
# You should have received a copy of the GNU Affero General Public License
# along with this program. If not, see <https://www.gnu.org/licenses/>.
import pandas as pd
import os
from itertools import chain
def filepathInitialGroup(df: pd.DataFrame):
original_columns = df.columns.tolist()
# Step 1: Split comma-separated filepaths into lists
df["filename_x"] = df["filename_x"].str.split(",")
# Step 2: Explode the list so each filepath becomes its own row
df = df.explode("filename_x", ignore_index=True)
# Step 3: Clean up whitespace and normalize paths
df["filename_x"] = df["filename_x"].str.strip()
df["filename_x"] = df["filename_x"].str.replace(r"\\\\", r"\\", regex=True)
df["filename_x"] = df["filename_x"].apply(lambda x: os.path.normpath(x) if pd.notna(x) else "")
# Step 4: Extract directory and filename from each filepath
df["directory"] = df["filename_x"].apply(lambda x: os.path.normpath(os.path.dirname(x)) if pd.notna(x) else "")
df["filename"] = df["filename_x"].apply(lambda x: os.path.basename(x) if pd.notna(x) else "")
# Step 5: Drop the original raw filepath column
df = df.drop(columns=["filename_x"])
# Helper functions for path manipulation
def get_parts(path):
return os.path.normpath(path).split(os.sep)
def join_parts(parts):
return os.path.normpath(os.sep.join(parts))
def longest_common_prefix(paths):
split_paths = [get_parts(p) for p in paths]
min_len = min(len(p) for p in split_paths)
prefix = []
for i in range(min_len):
current = split_paths[0][i]
if all(p[i] == current for p in split_paths):
prefix.append(current)
else:
break
return join_parts(prefix)
# Step 6: Group directories by shared prefix
directories = df["directory"].tolist()
groups = []
used = set()
for i, path in enumerate(directories):
if path in used:
continue
group = [path]
parts_i = get_parts(path)
for j in range(i + 1, len(directories)):
parts_j = get_parts(directories[j])
common = os.path.commonprefix([parts_i, parts_j])
if (len(parts_i) > 3 and len(common) >= 3) or (len(parts_i) == 3 and len(common) >= 2):
group.append(directories[j])
used.add(directories[j])
elif len(common) == len(parts_i) - 1 and len(parts_i) > 3:
group.append(directories[j])
used.add(directories[j])
used.add(path)
groups.append(group)
# Step 7: Map each original directory to its grouped prefix
prefix_map = {dir: longest_common_prefix(group) for group in groups for dir in group}
df["grouped_directory"] = df["directory"].map(prefix_map)
# Step 8: Group the DataFrame by grouped_directory
aggregation = {col: (lambda x: list(x)) for col in original_columns if col not in ["filename_x"]}
aggregation.update({
"directory": lambda x: list(x),
"filename": lambda x: list(x)
})
grouped_df = df.groupby("grouped_directory", as_index=False).agg(aggregation)
# Step 9: Split into eligible and ineligible paths based on depth
grouped_df["depth"] = grouped_df["grouped_directory"].apply(lambda x: len(get_parts(x)))
path_eligible = grouped_df[grouped_df["depth"] > 2].drop(columns=["depth"])
path_ineligible = grouped_df[grouped_df["depth"] <= 2].drop(columns=["depth"])
# Step 10: Move entries from eligible to ineligible if grouped_directory contains excluded directories
mask = path_eligible["grouped_directory"].str.contains(r"(?i)(?:\\Users|\\c\$\\Users|inetpub\\wwwroot|windows\\temp)", na=False)
move_to_ineligible = path_eligible[mask]
path_eligible = path_eligible[~mask]
path_ineligible = pd.concat([path_ineligible, move_to_ineligible], ignore_index=True)
# Step 11: Deduplicate list elements in all columns
def deduplicate_lists(df):
for col in df.columns:
if df[col].apply(lambda x: isinstance(x, list)).all():
df[col] = df[col].apply(lambda x: list({str(item): item for item in chain.from_iterable(x if isinstance(x[0], list) else [x])}.values()))
return df
path_eligible = deduplicate_lists(path_eligible)
path_ineligible = deduplicate_lists(path_ineligible)
return path_eligible, path_ineligible
def filter_and_drop(approved, eligiblepaths, min_hashes):
"""
Filters eligiblepaths to rows where all hashes are in approved,
then drops rows with fewer than min_hashes hashes.
"""
approved_hashes = set(approved['sha256'])
def all_hashes_approved(row):
return all(h in approved_hashes for h in row['sha256'])
filtered = eligiblepaths[eligiblepaths.apply(all_hashes_approved, axis=1)]
filtered = filtered[filtered['sha256'].apply(len) >= min_hashes]
return filtered