# Copyright (C) 2025 James Brotosky, Brandon Wickline # # This program is free software: you can redistribute it and/or modify # it under the terms of the GNU Affero General Public License as published # by the Free Software Foundation, either version 3 of the License, or # (at your option) any later version. # # This program is distributed in the hope that it will be useful, # but WITHOUT ANY WARRANTY; without even the implied warranty of # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the # GNU Affero General Public License for more details. # # You should have received a copy of the GNU Affero General Public License # along with this program. If not, see . import pandas as pd import os from itertools import chain def filepathInitialGroup(df: pd.DataFrame): original_columns = df.columns.tolist() # Step 1: Split comma-separated filepaths into lists df["filename_x"] = df["filename_x"].str.split(",") # Step 2: Explode the list so each filepath becomes its own row df = df.explode("filename_x", ignore_index=True) # Step 3: Clean up whitespace and normalize paths df["filename_x"] = df["filename_x"].str.strip() df["filename_x"] = df["filename_x"].str.replace(r"\\\\", r"\\", regex=True) df["filename_x"] = df["filename_x"].apply(lambda x: os.path.normpath(x) if pd.notna(x) else "") # Step 4: Extract directory and filename from each filepath df["directory"] = df["filename_x"].apply(lambda x: os.path.normpath(os.path.dirname(x)) if pd.notna(x) else "") df["filename"] = df["filename_x"].apply(lambda x: os.path.basename(x) if pd.notna(x) else "") # Step 5: Drop the original raw filepath column df = df.drop(columns=["filename_x"]) # Helper functions for path manipulation def get_parts(path): return os.path.normpath(path).split(os.sep) def join_parts(parts): return os.path.normpath(os.sep.join(parts)) def longest_common_prefix(paths): split_paths = [get_parts(p) for p in paths] min_len = min(len(p) for p in split_paths) prefix = [] for i in range(min_len): current = split_paths[0][i] if all(p[i] == current for p in split_paths): prefix.append(current) else: break return join_parts(prefix) # Step 6: Group directories by shared prefix directories = df["directory"].tolist() groups = [] used = set() for i, path in enumerate(directories): if path in used: continue group = [path] parts_i = get_parts(path) for j in range(i + 1, len(directories)): parts_j = get_parts(directories[j]) common = os.path.commonprefix([parts_i, parts_j]) if (len(parts_i) > 3 and len(common) >= 3) or (len(parts_i) == 3 and len(common) >= 2): group.append(directories[j]) used.add(directories[j]) elif len(common) == len(parts_i) - 1 and len(parts_i) > 3: group.append(directories[j]) used.add(directories[j]) used.add(path) groups.append(group) # Step 7: Map each original directory to its grouped prefix prefix_map = {dir: longest_common_prefix(group) for group in groups for dir in group} df["grouped_directory"] = df["directory"].map(prefix_map) # Step 8: Group the DataFrame by grouped_directory aggregation = {col: (lambda x: list(x)) for col in original_columns if col not in ["filename_x"]} aggregation.update({ "directory": lambda x: list(x), "filename": lambda x: list(x) }) grouped_df = df.groupby("grouped_directory", as_index=False).agg(aggregation) # Step 9: Split into eligible and ineligible paths based on depth grouped_df["depth"] = grouped_df["grouped_directory"].apply(lambda x: len(get_parts(x))) path_eligible = grouped_df[grouped_df["depth"] > 2].drop(columns=["depth"]) path_ineligible = grouped_df[grouped_df["depth"] <= 2].drop(columns=["depth"]) # Step 10: Move entries from eligible to ineligible if grouped_directory contains excluded directories mask = path_eligible["grouped_directory"].str.contains(r"(?i)(?:\\Users|\\c\$\\Users|inetpub\\wwwroot|windows\\temp)", na=False) move_to_ineligible = path_eligible[mask] path_eligible = path_eligible[~mask] path_ineligible = pd.concat([path_ineligible, move_to_ineligible], ignore_index=True) # Step 11: Deduplicate list elements in all columns def deduplicate_lists(df): for col in df.columns: if df[col].apply(lambda x: isinstance(x, list)).all(): df[col] = df[col].apply(lambda x: list({str(item): item for item in chain.from_iterable(x if isinstance(x[0], list) else [x])}.values())) return df path_eligible = deduplicate_lists(path_eligible) path_ineligible = deduplicate_lists(path_ineligible) return path_eligible, path_ineligible