Paths fixed... need to double check the path exclusion format to feed api

This commit is contained in:
=
2025-09-03 16:41:27 -04:00
parent b41e56d84a
commit 935467fbcb
3 changed files with 133 additions and 170 deletions
+57 -100
View File
@@ -18,6 +18,7 @@ import gc
import json import json
import os import os
import pandas as pd import pandas as pd
import re
import urllib3 import urllib3
import utils.allowlist import utils.allowlist
import utils.getdeviceevents import utils.getdeviceevents
@@ -26,13 +27,14 @@ import utils.pathfunctions
import utils.policyfunctions import utils.policyfunctions
import utils.pretty as ct import utils.pretty as ct
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning) urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
dotenv.load_dotenv() dotenv.load_dotenv()
#Constants #Constants
url = os.getenv('url') url = os.getenv('url')
badpublisherlist = ["Brave Software, Inc.", "Zoom Video Communications, Inc.", "GlavSoft LLC"] bad_publisher_list = ["Brave","Zoom", "GlavSoft", "VNC"]
pups = ["logmein", "invalid"] pups = ["logmein", "invalid"]
badpathparts = ["users", "wwwroot", "windows\\temp", "windows\\task", "windows\\system32", "startup", "windows\\fonts", "Recycle.Bin", "AppData", "programdata"] badpathparts = ["users", "wwwroot", "windows\\temp", "windows\\task", "windows\\system32", "startup", "windows\\fonts", "Recycle.Bin", "AppData", "programdata"]
path_exclusion_constant = 3 path_exclusion_constant = 3
@@ -277,7 +279,7 @@ def menu_prepare_to_enforce():
categorized = utils.hashfunctions.categorizeHashes( categorized = utils.hashfunctions.categorizeHashes(
pd.read_parquet(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"), pd.read_parquet(f"parquet\\combined_hashlist_{first_policy}_{second_policy}.parquet"),
threat_tolerance_constant, threat_tolerance_constant,
badpublisherlist, bad_publisher_list,
pups pups
) )
@@ -325,86 +327,15 @@ def menu_prepare_to_enforce():
del condensed_combo del condensed_combo
gc.collect() gc.collect()
if not os.path.exists(f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv"):
condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet")
needsapproval= pd.read_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet")
#Pull hash info for the entries in the needs approval table
needsapproval = pd.merge(condensed_combo, needsapproval, on='sha256', how='inner')
#Deduplicate lists in the columns
for col in needsapproval.columns:
if needsapproval[col].apply(lambda x: isinstance(x, list)).all():
needsapproval[col] = needsapproval[col].apply(deduplicate_list)
#Rename Publisher, Keep and reorder columns we want
needsapproval = needsapproval.rename(columns={'publisher_x': 'publisher'})
needsapproval = needsapproval[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']]
needsapproval['filename_key'] = needsapproval['filename'].apply(lambda x: x[0] if isinstance(x, list) and x else '')
needsapproval = needsapproval.sort_values(by='filename_key').drop(columns=['filename_key'])
needsapproval.to_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet",index=False)
del needsapproval
del condensed_combo
gc.collect()
if not os.path.exists(f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv"):
condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet")
needsapproval= pd.read_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet")
#Pull hash info for the entries in the needs approval table
needsapproval = pd.merge(condensed_combo, needsapproval, on='sha256', how='inner')
#Deduplicate lists in the columns
for col in needsapproval.columns:
if needsapproval[col].apply(lambda x: isinstance(x, list)).all():
needsapproval[col] = needsapproval[col].apply(deduplicate_list)
#Rename Publisher, Keep and reorder columns we want
needsapproval = needsapproval.rename(columns={'publisher_x': 'publisher'})
needsapproval = needsapproval[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']]
needsapproval['filename_key'] = needsapproval['filename'].apply(lambda x: x[0] if isinstance(x, list) and x else '')
needsapproval = needsapproval.sort_values(by='filename_key').drop(columns=['filename_key'])
needsapproval.to_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", index=False)
del needsapproval
del condensed_combo
gc.collect()
if not os.path.exists(f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.html"):
condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet")
needsapproval= pd.read_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet")
#Pull hash info for the entries in the needs approval table
needsapproval = pd.merge(condensed_combo, needsapproval, on='sha256', how='inner')
#Deduplicate lists in the columns
for col in needsapproval.columns:
if needsapproval[col].apply(lambda x: isinstance(x, list)).all():
needsapproval[col] = needsapproval[col].apply(deduplicate_list)
#Rename Publisher, Keep and reorder columns we want
needsapproval = needsapproval.rename(columns={'publisher_x': 'publisher'})
needsapproval = needsapproval[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']]
needsapproval['filename_key'] = needsapproval['filename'].apply(lambda x: x[0] if isinstance(x, list) and x else '')
needsapproval = needsapproval.sort_values(by='filename_key').drop(columns=['filename_key'])
needsapproval.to_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", index=False)
del needsapproval
del condensed_combo
gc.collect()
if os.path.exists(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") & os.path.exists(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") & os.path.exists(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet"): if os.path.exists(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") & os.path.exists(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") & os.path.exists(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet"):
utils.hashfunctions.combineHashAndHist(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet", first_policy, second_policy)
utils.hashfunctions.combineHashAndHist(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet", first_policy, second_policy)
utils.hashfunctions.combineHashAndHist(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet", first_policy, second_policy)
unknown = pd.read_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet") unknown = pd.read_parquet(f"parquet\\hashes_rep_unknown_{first_policy}_{second_policy}.parquet")
good = pd.read_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet") good = pd.read_parquet(f"parquet\\hashes_rep_good_{first_policy}_{second_policy}.parquet")
bad = pd.read_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet") bad = pd.read_parquet(f"parquet\\hashes_rep_bad_{first_policy}_{second_policy}.parquet")
@@ -423,15 +354,14 @@ def menu_prepare_to_enforce():
unknown = unknown[~unknown["filename"].str.contains(pattern, na=False)] unknown = unknown[~unknown["filename"].str.contains(pattern, na=False)]
good = good[~good["filename"].str.contains(pattern, na=False)] good = good[~good["filename"].str.contains(pattern, na=False)]
unknown.to_csv(f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv",index=False) unknown.to_csv(f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv",index=False)
good.to_csv(f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv",index=False) good.to_csv(f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.csv",index=False)
bad.to_csv(f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.csv",index=False)
ct.style_dataframe_dark(unknown, f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.html") ct.style_dataframe_dark(unknown, f"needs_approved\\hashes_rep_unknown_{first_policy}_{second_policy}.html")
ct.style_dataframe_dark(good, f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.html") ct.style_dataframe_dark(good, f"needs_approved\\hashes_rep_good_{first_policy}_{second_policy}.html")
ct.style_dataframe_dark(bad, f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.html") ct.style_dataframe_dark(bad, f"needs_approved\\hashes_rep_bad_{first_policy}_{second_policy}.html")
elif choice == "3": elif choice == "3":
if os.path.exists(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv") and os.path.exists(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv"): if os.path.exists(f"approved\\hashes_rep_unknown_{first_policy}_{second_policy}.csv") and os.path.exists(f"approved\\hashes_rep_good_{first_policy}_{second_policy}.csv"):
@@ -454,54 +384,67 @@ def menu_prepare_to_enforce():
if not os.path.exists(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet"): if not os.path.exists(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet"):
all_approved_hashes = pd.read_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet") all_approved_hashes = pd.read_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet")
print(ct.colorText(f"Beginning calculating longest common filepaths for path exceptions","green")) print(ct.colorText(f"Beginning calculating longest common filepaths for path exceptions","green"))
grouped_df_view, df_with_groups_appended = utils.pathfunctions.export_groups_for_review(all_approved_hashes,"filename","longestcfp",min_files_for_path,path_exclusion_constant)
df_with_groups_appended.to_parquet(f"parquet\\approved_hashes_with_paths_{first_policy}_{second_policy}.parquet", index=False) haslcp = utils.pathfunctions.split_filepaths_grouped(all_approved_hashes)
haslcp.drop_duplicates()
forbidden = utils.pathfunctions.regulator(badpathparts, True) forbidden = utils.pathfunctions.regulator(badpathparts, True)
forbidden_lcfp = grouped_df_view["longestcfp"].str.contains(forbidden, na=False) forbidden_lcfp = haslcp["longestcfp"].str.contains(forbidden, na=False)
# Make a real DataFrame copy before modifying
grouped_df_view = grouped_df_view[~forbidden_lcfp].copy()
print(ct.colorText("Removing forbidden filepaths for path exceptions", "green")) print(ct.colorText("Removing forbidden filepaths for path exceptions", "green"))
for col in grouped_df_view.columns: # Make a real DataFrame copy before modifying
if grouped_df_view[col].apply(lambda x: isinstance(x, list)).all(): lcp_not_forbidden = haslcp[~forbidden_lcfp].copy()
grouped_df_view[col] = grouped_df_view[col].apply(deduplicate_list)
grouped_df_view.to_parquet(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet", index=False) #For the review, drop down to only the columns we care, and then group by the commmon file path, consolidating and dropping dupes
grouped_df_view.to_csv(f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.csv", index=False) lcp_not_forbidden_review = lcp_not_forbidden[['longestcfp', 'middle', 'filename_only', 'sha256']]
ct.style_dataframe_dark(grouped_df_view, f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.html")
# Count unique sha256 per longestcfp
unique_sha_counts = lcp_not_forbidden_review.groupby('longestcfp')['sha256'].nunique().reset_index()
unique_sha_counts.columns = ['longestcfp', 'unique_sha256_count']
# Merge the count back into the original DataFrame
lcp_not_forbidden_review = lcp_not_forbidden_review.merge(unique_sha_counts, on='longestcfp', how='left')
lcp_not_forbidden_review = lcp_not_forbidden_review[lcp_not_forbidden_review['unique_sha256_count'] >= min_files_for_path]
lcp_not_forbidden_review.to_parquet(f"parquet\\path_needs_approved_{first_policy}_{second_policy}.parquet",index=False)
lcp_not_forbidden_review.to_csv(f"needs_approved\\path_needs_approved_{first_policy}_{second_policy}.csv",index=False)
del grouped_df_view
del df_with_groups_appended
gc.collect()
else: else:
print(ct.colorText(f"Please manually approve hashes prior to this step","red")) print(ct.colorText(f"Please manually approve hashes prior to this step","red"))
elif choice == "4": elif choice == "4":
if os.path.exists(f"parquet\\approved_hashes_with_paths_{first_policy}_{second_policy}.parquet") and os.path.exists(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv"): if os.path.exists(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv"):
if not os.path.exists(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet"): if not os.path.exists(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet") and not os.path.exists(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet"):
df1 = pd.read_parquet(f"parquet\\approved_hashes_with_paths_{first_policy}_{second_policy}.parquet") allhashes = pd.read_parquet(f"parquet\\all_approved_hashes_{first_policy}_{second_policy}.parquet")
allowbyhash = utils.pathfunctions.mask_from_csv(df1, f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv","longestcfp")
pathexclusions = tryToReadCSV(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv") pathexclusions = tryToReadCSV(f"approved\\path_needs_approved_{first_policy}_{second_policy}.csv")
pathexclusions.to_parquet(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet", index=False) pathexclusions.to_parquet(f"parquet\\final_path_exclusions_{first_policy}_{second_policy}.parquet", index=False)
allowbyhash = allhashes[~allhashes['sha256'].isin(pathexclusions['sha256'])]
allowbyhash.to_parquet(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet", index=False) allowbyhash.to_parquet(f"parquet\\final_hash_approvals_{first_policy}_{second_policy}.parquet", index=False)
easyview = allowbyhash.groupby('sha256').agg(list).reset_index() easyview = allowbyhash.groupby('sha256').agg(list).reset_index()
# Deduplicate all list columns in easyview
for col in easyview.columns:
if col != 'sha256': # Skip the grouping column
easyview[col] = easyview[col].apply(lambda x: list(set(x)))
easyview = easyview.sort_values(by=["reputation_status", "filename"])
ct.style_dataframe_dark(easyview, f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html") ct.style_dataframe_dark(easyview, f"preflight\\final_hash_approvals_{first_policy}_{second_policy}.html")
ct.style_dataframe_dark(pathexclusions, f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html") ct.style_dataframe_dark(pathexclusions, f"preflight\\final_path_exclusions_{first_policy}_{second_policy}.html")
del df1
del allowbyhash del allowbyhash
del pathexclusions del pathexclusions
del easyview del easyview
@@ -542,7 +485,19 @@ def menu_prepare_to_enforce():
print(ct.colorText("Proceeding with the code...", "yellow")) print(ct.colorText("Proceeding with the code...", "yellow"))
print(ct.colorText(f"Adding path exclusions to {destination_name}", "yellow")) print(ct.colorText(f"Adding path exclusions to {destination_name}", "yellow"))
pathexcludelist = pathexclusions['longestcfp'].unique().tolist() pathexcludelist = pathexclusions['longestcfp'].unique().tolist()
utils.policyfunctions.addPath(destination_id,pathexcludelist)
# Regex to match a Windows drive letter at the start (e.g., C:\)
drive_letter_pattern = re.compile(r'^[a-zA-Z]:\\')
# Processed list
processed_paths = [
(path if drive_letter_pattern.match(path) else f"\\\\{path}") + "**"
for path in pathexcludelist
]
utils.policyfunctions.addPath(destination_id,processed_paths)
print(ct.colorText(f"Adding hashes to {allowlist_parent_name}", "yellow")) print(ct.colorText(f"Adding hashes to {allowlist_parent_name}", "yellow"))
@@ -554,6 +509,8 @@ def menu_prepare_to_enforce():
utils.policyfunctions.addHash(allowlist_child_id, allowlist_childhashlist) utils.policyfunctions.addHash(allowlist_child_id, allowlist_childhashlist)
ct.locked() ct.locked()
print(repr(processed_paths))
print(processed_paths)
exit() exit()
else: else:
+22 -3
View File
@@ -19,6 +19,7 @@ import os
import json import json
import utils.pathfunctions as pathf import utils.pathfunctions as pathf
import utils.pretty as ct import utils.pretty as ct
import gc
def aggregateHashes(executions_json) -> pd.DataFrame: def aggregateHashes(executions_json) -> pd.DataFrame:
@@ -103,7 +104,7 @@ def augmentAggregatedHashes(url, agg_df: pd.DataFrame) -> pd.DataFrame:
return aug_df return aug_df
def categorizeHashes(df: pd.DataFrame, threat_tolerance: int, untrusted_publishers: list, pups: list): def categorizeHashes(df: pd.DataFrame, threat_tolerance: int, untrusted_publishers, pups: list):
if untrusted_publishers is None: untrusted_publishers = [] if untrusted_publishers is None: untrusted_publishers = []
if pups is None: pups = [] if pups is None: pups = []
@@ -126,14 +127,14 @@ def categorizeHashes(df: pd.DataFrame, threat_tolerance: int, untrusted_publishe
mask_approved = ( mask_approved = (
( (
(df["publisher"] != "Not Signed") & (df["publisher"] != "Not Signed") &
~df["publisher"].isin(untrusted_publishers) & ~df["publisher"].str.contains(pathf.regulator(untrusted_publishers), case=False, na=False) &
~df["reputation_status"].isna() & ~df["reputation_status"].isna() &
~df["description"].str.contains(pathf.regulator(pups), case=False, na=False) ~df["description"].str.contains(pathf.regulator(pups), case=False, na=False)
) | ) |
( (
(df["publisher"] == "Not Signed") & (df["publisher"] == "Not Signed") &
~df["reputation_flag"] & ~df["reputation_flag"] &
~df["publisher"].isin(untrusted_publishers) & ~df["publisher"].str.contains(pathf.regulator(untrusted_publishers), case=False, na=False) &
~df["reputation_status"].isna() & ~df["reputation_status"].isna() &
~df["description"].str.contains(pathf.regulator(pups), case=False, na=False) ~df["description"].str.contains(pathf.regulator(pups), case=False, na=False)
) )
@@ -203,3 +204,21 @@ def destinationHashes(
# Concatenate results # Concatenate results
df_hashdestination = pd.concat([df_paths, df_hashes], ignore_index=True) df_hashdestination = pd.concat([df_paths, df_hashes], ignore_index=True)
return df_hashdestination return df_hashdestination
def combineHashAndHist(path, first_policy, second_policy):
condensed_combo = pd.read_parquet(f"parquet\\condensed_executions_{first_policy}_{second_policy}.parquet")
df = pd.read_parquet(path)
#Pull hash info for the entries in the needs approval table
df = pd.merge(condensed_combo, df, on='sha256', how='inner')
#Rename Publisher, Keep and reorder columns we want
df = df.rename(columns={'publisher_x': 'publisher'})
df = df[['sha256', 'publisher', 'description', 'filename', 'hostname', 'username', 'productname', 'productversion','reputation_lastseen', 'reputation_scannermatch', 'reputation_scannercount','reputation_status', 'reputation_threatlevel', 'reputation_threatname','reputation_timestamp', 'pprocess', 'gprocess', 'commandline']]
df = df.sort_values(by='filename')
df.to_parquet(path, index=False)
del df
del condensed_combo
gc.collect()
+44 -57
View File
@@ -15,73 +15,60 @@
import pandas as pd import pandas as pd
import os import os
from itertools import chain
import ast import ast
import re import re
def split_path(path): import os
parts = [] import pandas as pd
while True:
head, tail = os.path.split(path) import os
if tail: import pandas as pd
parts.insert(0, tail)
path = head def split_filepaths_grouped(df, col="filename", group_parts=3, min_parts=3):
else: def clean_split(path):
if head: parts = os.path.normpath(path).split(os.sep)
parts.insert(0, head) # Remove leading empty strings caused by UNC paths
break parts = [p for p in parts if p]
return parts return parts
def local_common_pass(paths, min_parts=3): df = df.copy()
results = {} split_paths = df[col].apply(clean_split)
paths_sorted = sorted(paths)
for i, path in enumerate(paths_sorted):
candidates = []
if i > 0: # Filter out paths with fewer than `min_parts` components
try: df = df[split_paths.apply(lambda parts: len(parts) >= min_parts)].copy()
candidates.append(os.path.commonpath([path, paths_sorted[i-1]])) split_paths = split_paths[df.index] # Update split_paths to match filtered df
except ValueError:
# different drives, skip
pass
if i < len(paths_sorted) - 1:
try:
candidates.append(os.path.commonpath([path, paths_sorted[i+1]]))
except ValueError:
# different drives, skip
pass
best = path df["group_key"] = split_paths.apply(lambda parts: os.sep.join(parts[:group_parts]))
best_len = 0 grouped = df.groupby("group_key")
for c in candidates: new_rows = []
parts = split_path(c)
if len(parts) >= min_parts and len(parts) > best_len:
best = c
best_len = len(parts)
results[path] = best
return results
def add_longest_common_two_local(df, col="filename_x", new_col="longestcfp", min_parts=3): for _, group_df in grouped:
dirs_series = df[col].astype(str).apply(os.path.dirname) paths = group_df[col].tolist()
first_pass = local_common_pass(dirs_series.tolist(), min_parts) split_parts = [clean_split(p) for p in paths]
second_pass = local_common_pass(list(first_pass.values()), min_parts)
df[new_col] = dirs_series.map(lambda d: second_pass[first_pass[d]])
return df
def export_groups_for_review(df, col, group_col, min_number_in_group, path_length_constant): def longest_common_prefix(paths):
""" if not paths:
Compute longest common paths, group filepaths, write CSV for review. return []
""" prefix = paths[0]
df = df.drop_duplicates(subset=[col], keep='first') for path in paths[1:]:
df = add_longest_common_two_local(df, col=col, new_col=group_col) prefix = [a for a, b in zip(prefix, path) if a == b]
grouped = df.groupby(group_col)[col].apply(list).reset_index() if not prefix:
grouped = grouped.sort_values(by=col) break
print("Before filtering:", len(grouped)) return prefix
grouped = grouped[grouped[col].apply(lambda x: len(x) >= min_number_in_group)]
filtered = grouped[grouped[group_col].apply(lambda x: len(os.path.normpath(x).split(os.sep)) >= path_length_constant)]
print("After filtering:", len(grouped))
return filtered, df common_prefix = longest_common_prefix(split_parts)
prefix_str = os.sep.join(common_prefix)
for i, parts in enumerate(split_parts):
filename = parts[-1]
middle = os.sep.join(parts[len(common_prefix):-1]) if len(parts) > len(common_prefix) + 1 else ""
row = group_df.iloc[i].copy()
row["longestcfp"] = prefix_str
row["middle"] = middle
row["filename_only"] = filename
new_rows.append(row)
return pd.DataFrame(new_rows).drop(columns=["group_key"])
def mask_from_csv(df, csv_path, filepath_col): def mask_from_csv(df, csv_path, filepath_col):
""" """