From c1246f4f657e19829ca1d1e977e906e7173e9b04 Mon Sep 17 00:00:00 2001 From: = <=> Date: Wed, 27 Aug 2025 09:35:21 -0400 Subject: [PATCH] recommit --- utils/pathfunctions.py | 157 ++++++++++++++++++----------------------- 1 file changed, 70 insertions(+), 87 deletions(-) diff --git a/utils/pathfunctions.py b/utils/pathfunctions.py index a5953f6..82fa278 100644 --- a/utils/pathfunctions.py +++ b/utils/pathfunctions.py @@ -16,106 +16,89 @@ import pandas as pd import os from itertools import chain +import ast -def filepathInitialGroup(df: pd.DataFrame): - original_columns = df.columns.tolist() - # Step 1: Split comma-separated filepaths into lists - df["filename_x"] = df["filename_x"].str.split(",") - # Step 2: Explode the list so each filepath becomes its own row - df = df.explode("filename_x", ignore_index=True) - # Step 3: Clean up whitespace and normalize paths - df["filename_x"] = df["filename_x"].str.strip() - df["filename_x"] = df["filename_x"].str.replace(r"\\\\", r"\\", regex=True) - df["filename_x"] = df["filename_x"].apply(lambda x: os.path.normpath(x) if pd.notna(x) else "") +def split_path(path): + parts = [] + while True: + head, tail = os.path.split(path) + if tail: + parts.insert(0, tail) + path = head + else: + if head: + parts.insert(0, head) + break + return parts - # Step 4: Extract directory and filename from each filepath - df["directory"] = df["filename_x"].apply(lambda x: os.path.normpath(os.path.dirname(x)) if pd.notna(x) else "") - df["filename"] = df["filename_x"].apply(lambda x: os.path.basename(x) if pd.notna(x) else "") +def local_common_pass(paths, min_parts=3): + results = {} + paths_sorted = sorted(paths) + for i, path in enumerate(paths_sorted): + candidates = [] - # Step 5: Drop the original raw filepath column - df = df.drop(columns=["filename_x"]) + if i > 0: + try: + candidates.append(os.path.commonpath([path, paths_sorted[i-1]])) + except ValueError: + # different drives, skip + pass + if i < len(paths_sorted) - 1: + try: + candidates.append(os.path.commonpath([path, paths_sorted[i+1]])) + except ValueError: + # different drives, skip + pass - # Helper functions for path manipulation - def get_parts(path): - return os.path.normpath(path).split(os.sep) + best = path + best_len = 0 + for c in candidates: + parts = split_path(c) + if len(parts) >= min_parts and len(parts) > best_len: + best = c + best_len = len(parts) + results[path] = best + return results - def join_parts(parts): - return os.path.normpath(os.sep.join(parts)) - def longest_common_prefix(paths): - split_paths = [get_parts(p) for p in paths] - min_len = min(len(p) for p in split_paths) - prefix = [] - for i in range(min_len): - current = split_paths[0][i] - if all(p[i] == current for p in split_paths): - prefix.append(current) - else: - break - return join_parts(prefix) +def add_longest_common_two_local(df, col="filename_x", new_col="longestcfp", min_parts=3): + dirs_series = df[col].astype(str).apply(os.path.dirname) + first_pass = local_common_pass(dirs_series.tolist(), min_parts) + second_pass = local_common_pass(list(first_pass.values()), min_parts) + df[new_col] = dirs_series.map(lambda d: second_pass[first_pass[d]]) + return df - # Step 6: Group directories by shared prefix - directories = df["directory"].tolist() - groups = [] - used = set() - for i, path in enumerate(directories): - if path in used: - continue - group = [path] - parts_i = get_parts(path) +def export_groups_for_review(df, col="filename_x", group_col="longestcfp", csv_path="filegroups_review.csv"): + """ + Compute longest common paths, group filepaths, write CSV for review. + """ + df = add_longest_common_two_local(df, col=col, new_col=group_col) + grouped = df.groupby(group_col)[col].apply(list).reset_index() + grouped.to_csv(csv_path, index=False) + print(f"Grouped file list saved to: {csv_path}") + return grouped, df - for j in range(i + 1, len(directories)): - parts_j = get_parts(directories[j]) - common = os.path.commonprefix([parts_i, parts_j]) - if (len(parts_i) > 3 and len(common) >= 3) or (len(parts_i) == 3 and len(common) >= 2): - group.append(directories[j]) - used.add(directories[j]) - elif len(common) == len(parts_i) - 1 and len(parts_i) > 3: - group.append(directories[j]) - used.add(directories[j]) - used.add(path) - groups.append(group) - - # Step 7: Map each original directory to its grouped prefix - prefix_map = {dir: longest_common_prefix(group) for group in groups for dir in group} - df["grouped_directory"] = df["directory"].map(prefix_map) - - # Step 8: Group the DataFrame by grouped_directory - aggregation = {col: (lambda x: list(x)) for col in original_columns if col not in ["filename_x"]} - aggregation.update({ - "directory": lambda x: list(x), - "filename": lambda x: list(x) - }) - - grouped_df = df.groupby("grouped_directory", as_index=False).agg(aggregation) - - # Step 9: Split into eligible and ineligible paths based on depth - grouped_df["depth"] = grouped_df["grouped_directory"].apply(lambda x: len(get_parts(x))) - path_eligible = grouped_df[grouped_df["depth"] > 2].drop(columns=["depth"]) - path_ineligible = grouped_df[grouped_df["depth"] <= 2].drop(columns=["depth"]) - - # Step 10: Move entries from eligible to ineligible if grouped_directory contains excluded directories - mask = path_eligible["grouped_directory"].str.contains(r"(?i)(?:\\Users|\\c\$\\Users|inetpub\\wwwroot|windows\\temp)", na=False) - move_to_ineligible = path_eligible[mask] - path_eligible = path_eligible[~mask] - path_ineligible = pd.concat([path_ineligible, move_to_ineligible], ignore_index=True) - - # Step 11: Deduplicate list elements in all columns - def deduplicate_lists(df): - for col in df.columns: - if df[col].apply(lambda x: isinstance(x, list)).all(): - df[col] = df[col].apply(lambda x: list({str(item): item for item in chain.from_iterable(x if isinstance(x[0], list) else [x])}.values())) - return df - - path_eligible = deduplicate_lists(path_eligible) - path_ineligible = deduplicate_lists(path_ineligible) - - return path_eligible, path_ineligible +def mask_from_csv(df, csv_path, filepath_col="filename_x", group_col="longestcfp"): + """ + Reads reviewed CSV of groups, keeps only files in approved groups. + """ + review_df = pd.read_csv(csv_path) + # Convert string representation of lists back to actual lists + review_df[filepath_col] = review_df[filepath_col].apply(ast.literal_eval) + + # Flatten all approved file paths into a set for masking + approved_files = set() + for paths in review_df[filepath_col]: + approved_files.update(paths) + + # Keep only rows in df that are in approved_files + masked_df = df[df[filepath_col].isin(approved_files)].copy() + return masked_df def filter_and_drop(approved, eligiblepaths, min_hashes): """