From 66e484101b4616ba968015a46cda9801548f568c Mon Sep 17 00:00:00 2001 From: Yuante Li <104308117+WinstonLiyt@users.noreply.github.com> Date: Wed, 2 Apr 2025 19:38:43 +0800 Subject: [PATCH] feat: add describe_data_folder_v2 (#738) * add describe_data_folder_v2 * fix ci * fix * add packages --- .../scenarios/data_science/scen/__init__.py | 217 +--------- rdagent/scenarios/data_science/scen/utils.py | 382 ++++++++++++++++++ requirements.txt | 4 + 3 files changed, 393 insertions(+), 210 deletions(-) create mode 100644 rdagent/scenarios/data_science/scen/utils.py diff --git a/rdagent/scenarios/data_science/scen/__init__.py b/rdagent/scenarios/data_science/scen/__init__.py index 7164914a..ca437806 100644 --- a/rdagent/scenarios/data_science/scen/__init__.py +++ b/rdagent/scenarios/data_science/scen/__init__.py @@ -1,17 +1,17 @@ import json -import os from pathlib import Path from typing import Dict -import pandas as pd -from PIL import Image, TiffTags - from rdagent.app.data_science.conf import DS_RD_SETTING from rdagent.components.coder.data_science.conf import get_ds_env from rdagent.core.experiment import FBWorkspace from rdagent.core.scenario import Scenario from rdagent.log import rdagent_logger as logger from rdagent.oai.llm_utils import APIBackend +from rdagent.scenarios.data_science.scen.utils import ( + describe_data_folder, + describe_data_folder_v2, +) from rdagent.scenarios.kaggle.kaggle_crawler import ( crawl_descriptions, leaderboard_scores, @@ -19,211 +19,6 @@ from rdagent.scenarios.kaggle.kaggle_crawler import ( from rdagent.utils.agent.tpl import T -def read_csv_head(file_path, indent=0, lines=5, max_col_width=100): - """ - Reads the first few rows of a CSV file and formats them with indentation and optional truncation. - - Parameters: - file_path (str): Path to the CSV file. - indent (int): Number of spaces to prepend to each line for indentation. - lines (int): Number of rows to read from the CSV file. - max_col_width (int): Maximum width of each column's content. - - Returns: - str: A formatted string of the first few rows of the CSV file. - """ - try: - # Read the CSV file with specified rows - df = pd.read_csv(file_path, nrows=lines) - - if df.empty: - return " " * indent + "(No data in the file)" - - # Truncate column contents to a maximum width - truncated_df = df.copy() - for col in truncated_df.columns: - truncated_df[col] = ( - truncated_df[col] - .astype(str) - .apply(lambda x: (x[:max_col_width] + "...") if len(x) > max_col_width else x) - ) - - # Convert DataFrame to a string representation - df_string_lines = truncated_df.to_string(index=False).split("\n") - - # Add indentation to each line - indented_lines = [" " * indent + line for line in df_string_lines] - - return "\n".join(indented_lines) - except FileNotFoundError: - return f"Error: File not found at path '{file_path}'." - except pd.errors.EmptyDataError: - return f"Error: The file at '{file_path}' is empty." - except Exception as e: - return f"Error reading CSV: {e}" - - -def get_dir_snapshot(folder_path): - """ - [note] - - Returns a set of file extensions within the subfolder (excluding subfolder names) - - Compares only the types of files contained, not specific file names or quantities - """ - exts = set() - try: - with os.scandir(folder_path) as it: - for entry in it: - if entry.is_file(): - file_ext = os.path.splitext(entry.name)[1] - exts.add(file_ext) - except Exception as e: - logger.error(f"Error scanning directory: {e}") - - return frozenset(exts) - - -def describe_data_folder(folder_path, indent=0, max_files=2, partial_expand_subfolders=2, is_top_level=True): - """ - folder_path : Current directory path - indent : Current indentation - max_files : Maximum number of files of the same type to display - partial_expand_subfolders: When all subfolders have the same internal file types, only expand this many subfolders, the rest are omitted - is_top_level : Indicates if the current folder is the top-level folder - """ - result = [] - files_count = {} - files_details = {} - - for root, dirs, files in os.walk(folder_path): - dirs.sort() - files.sort() - if not dirs: - for file in files: - file_path = os.path.join(root, file) - file_type = os.path.splitext(file)[1][1:] - file_size = os.path.getsize(file_path) - - if file_type not in files_count: - files_count[file_type] = 0 - files_details[file_type] = [] - files_count[file_type] += 1 - - # At top level, collect all CSV and Markdown files without restrictions - # In deeper levels, follow the max_files restriction - if is_top_level and file_type in ["csv", "md"]: - files_details[file_type].append((file, file_size, file_path)) - elif len(files_details[file_type]) < max_files: - files_details[file_type].append((file, file_size, file_path)) - break - - # Collect "type snapshots" of subfolders - snapshots = [] - for d in dirs: - subfolder_path = os.path.join(root, d) - snapshot = get_dir_snapshot(subfolder_path) - snapshots.append(snapshot) - - # Determine if all subfolders have the same file type distribution - first_snapshot = snapshots[0] - all_same_structure = all(s == first_snapshot for s in snapshots) - - if all_same_structure: - for i, d in enumerate(dirs): - if i < partial_expand_subfolders: - result.append(" " * indent + f"- Folder: {d}") - subfolder_path = os.path.join(root, d) - result.append( - describe_data_folder( - folder_path=subfolder_path, - indent=indent + 2, - max_files=max_files, - partial_expand_subfolders=partial_expand_subfolders, - is_top_level=False, - ) - ) - else: - remaining = len(dirs) - i - result.append(" " * indent + f"... ({remaining} more subfolders)") - break - else: - for d in dirs: - result.append(" " * indent + f"- Folder: {d}") - subfolder_path = os.path.join(root, d) - result.append( - describe_data_folder( - folder_path=subfolder_path, - indent=indent + 2, - max_files=max_files, - partial_expand_subfolders=partial_expand_subfolders, - is_top_level=False, - ) - ) - - for file in files: - file_path = os.path.join(root, file) - file_type = os.path.splitext(file)[1][1:] - file_size = os.path.getsize(file_path) - - if file_type not in files_count: - files_count[file_type] = 0 - files_details[file_type] = [] - files_count[file_type] += 1 - - # At top level, collect all CSV and Markdown files without restrictions - # In deeper levels, follow the max_files restriction - if is_top_level and file_type in ["csv", "md"]: - files_details[file_type].append((file, file_size, file_path)) - elif not is_top_level and len(files_details[file_type]) <= max_files: - files_details[file_type].append((file, file_size, file_path)) - - break - - # Print the folder and its contents - for file_type, count in files_count.items(): - if count > max_files and file_type not in ["csv", "md", "txt"]: - result.append(" " * indent + f"{count} {file_type}s:") - for file, size, path in files_details[file_type]: - result.append(" " * (indent + 2) + f"- {file} ({size} bytes)") - result.append(" " * (indent + 2) + "... (file limit reached)") - else: - for file, size, path in files_details[file_type]: - if file_type == "csv": - df = pd.read_csv(path) - result.append( - " " * indent + f"- {file} ({size} bytes, with {df.shape[0]} rows and {df.shape[1]} columns)" - ) - result.append(" " * (indent + 2) + f"- Head of {file}:") - csv_head = read_csv_head(path, indent + 4) - result.append(csv_head) - continue - result.append(" " * indent + f"- {file} ({size} bytes)") - if file_type == "md": - result.append(" " * (indent + 2) + f"- Content of {file}:") - if file == "description.md": - result.append(" " * (indent + 4) + f"Please refer to the background of the scenario context.") - continue - with open(path, "r", encoding="utf-8") as f: - result.append(" " * (indent + 4) + f.read()) - if file_type == "tif": - result.append(" " * (indent + 2) + f"- Metadata of {file}:") - with Image.open(path) as img: - for tag, value in img.tag_v2.items(): - tag_name = TiffTags.TAGS_V2.get(tag, f"Unknown Tag {tag}") - result.append(" " * (indent + 4) + f"{tag_name}: {value}") - if file_type in ["json", "txt"]: - result.append(" " * (indent + 2) + f"- Content of {file}:") - with open(path, "r", encoding="utf-8") as f: - for i, line in enumerate(f): - if i < 2: - result.append( - " " * (indent + 4) + line.strip()[:100] + ("..." if len(line.strip()) > 100 else "") - ) - else: - break - - return "\n".join(result) + "\n" - - class DataScienceScen(Scenario): """Data Science Scenario""" @@ -333,7 +128,7 @@ class DataScienceScen(Scenario): return stdout def _get_data_folder_description(self) -> str: - return describe_data_folder(Path(DS_RD_SETTING.local_data_path) / self.competition) + return describe_data_folder_v2(Path(DS_RD_SETTING.local_data_path) / self.competition) class KaggleScen(DataScienceScen): @@ -364,3 +159,5 @@ class KaggleScen(DataScienceScen): if __name__ == "__main__": print(describe_data_folder(Path("/data/userdata/share/mle_kaggle") / "stanford-covid-vaccine")) + + print(describe_data_folder_v2(Path("/data/userdata/share/mle_kaggle") / "stanford-covid-vaccine")) diff --git a/rdagent/scenarios/data_science/scen/utils.py b/rdagent/scenarios/data_science/scen/utils.py new file mode 100644 index 00000000..f5d3a3e5 --- /dev/null +++ b/rdagent/scenarios/data_science/scen/utils.py @@ -0,0 +1,382 @@ +import os + +import pandas as pd +from PIL import Image, TiffTags + +from rdagent.log import rdagent_logger as logger + +""" data folder description version 1 """ + + +def read_csv_head(file_path, indent=0, lines=5, max_col_width=100): + """ + Reads the first few rows of a CSV file and formats them with indentation and optional truncation. + + Parameters: + file_path (str): Path to the CSV file. + indent (int): Number of spaces to prepend to each line for indentation. + lines (int): Number of rows to read from the CSV file. + max_col_width (int): Maximum width of each column's content. + + Returns: + str: A formatted string of the first few rows of the CSV file. + """ + try: + # Read the CSV file with specified rows + df = pd.read_csv(file_path, nrows=lines) + + if df.empty: + return " " * indent + "(No data in the file)" + + # Truncate column contents to a maximum width + truncated_df = df.copy() + for col in truncated_df.columns: + truncated_df[col] = ( + truncated_df[col] + .astype(str) + .apply(lambda x: (x[:max_col_width] + "...") if len(x) > max_col_width else x) + ) + + # Convert DataFrame to a string representation + df_string_lines = truncated_df.to_string(index=False).split("\n") + + # Add indentation to each line + indented_lines = [" " * indent + line for line in df_string_lines] + + return "\n".join(indented_lines) + except FileNotFoundError: + return f"Error: File not found at path '{file_path}'." + except pd.errors.EmptyDataError: + return f"Error: The file at '{file_path}' is empty." + except Exception as e: + return f"Error reading CSV: {e}" + + +def get_dir_snapshot(folder_path): + """ + [note] + - Returns a set of file extensions within the subfolder (excluding subfolder names) + - Compares only the types of files contained, not specific file names or quantities + """ + exts = set() + try: + with os.scandir(folder_path) as it: + for entry in it: + if entry.is_file(): + file_ext = os.path.splitext(entry.name)[1] + exts.add(file_ext) + except Exception as e: + logger.error(f"Error scanning directory: {e}") + + return frozenset(exts) + + +def describe_data_folder(folder_path, indent=0, max_files=2, partial_expand_subfolders=2, is_top_level=True): + """ + folder_path : Current directory path + indent : Current indentation + max_files : Maximum number of files of the same type to display + partial_expand_subfolders: When all subfolders have the same internal file types, only expand this many subfolders, the rest are omitted + is_top_level : Indicates if the current folder is the top-level folder + """ + result = [] + files_count = {} + files_details = {} + + for root, dirs, files in os.walk(folder_path): + dirs.sort() + files.sort() + if not dirs: + for file in files: + file_path = os.path.join(root, file) + file_type = os.path.splitext(file)[1][1:] + file_size = os.path.getsize(file_path) + + if file_type not in files_count: + files_count[file_type] = 0 + files_details[file_type] = [] + files_count[file_type] += 1 + + # At top level, collect all CSV and Markdown files without restrictions + # In deeper levels, follow the max_files restriction + if is_top_level and file_type in ["csv", "md"]: + files_details[file_type].append((file, file_size, file_path)) + elif len(files_details[file_type]) < max_files: + files_details[file_type].append((file, file_size, file_path)) + break + + # Collect "type snapshots" of subfolders + snapshots = [] + for d in dirs: + subfolder_path = os.path.join(root, d) + snapshot = get_dir_snapshot(subfolder_path) + snapshots.append(snapshot) + + # Determine if all subfolders have the same file type distribution + first_snapshot = snapshots[0] + all_same_structure = all(s == first_snapshot for s in snapshots) + + if all_same_structure: + for i, d in enumerate(dirs): + if i < partial_expand_subfolders: + result.append(" " * indent + f"- Folder: {d}") + subfolder_path = os.path.join(root, d) + result.append( + describe_data_folder( + folder_path=subfolder_path, + indent=indent + 2, + max_files=max_files, + partial_expand_subfolders=partial_expand_subfolders, + is_top_level=False, + ) + ) + else: + remaining = len(dirs) - i + result.append(" " * indent + f"... ({remaining} more subfolders)") + break + else: + for d in dirs: + result.append(" " * indent + f"- Folder: {d}") + subfolder_path = os.path.join(root, d) + result.append( + describe_data_folder( + folder_path=subfolder_path, + indent=indent + 2, + max_files=max_files, + partial_expand_subfolders=partial_expand_subfolders, + is_top_level=False, + ) + ) + + for file in files: + file_path = os.path.join(root, file) + file_type = os.path.splitext(file)[1][1:] + file_size = os.path.getsize(file_path) + + if file_type not in files_count: + files_count[file_type] = 0 + files_details[file_type] = [] + files_count[file_type] += 1 + + # At top level, collect all CSV and Markdown files without restrictions + # In deeper levels, follow the max_files restriction + if is_top_level and file_type in ["csv", "md"]: + files_details[file_type].append((file, file_size, file_path)) + elif not is_top_level and len(files_details[file_type]) <= max_files: + files_details[file_type].append((file, file_size, file_path)) + + break + + # Print the folder and its contents + for file_type, count in files_count.items(): + if count > max_files and file_type not in ["csv", "md", "txt"]: + result.append(" " * indent + f"{count} {file_type}s:") + for file, size, path in files_details[file_type]: + result.append(" " * (indent + 2) + f"- {file} ({size} bytes)") + result.append(" " * (indent + 2) + "... (file limit reached)") + else: + for file, size, path in files_details[file_type]: + if file_type == "csv": + df = pd.read_csv(path) + result.append( + " " * indent + f"- {file} ({size} bytes, with {df.shape[0]} rows and {df.shape[1]} columns)" + ) + result.append(" " * (indent + 2) + f"- Head of {file}:") + csv_head = read_csv_head(path, indent + 4) + result.append(csv_head) + continue + result.append(" " * indent + f"- {file} ({size} bytes)") + if file_type == "md": + result.append(" " * (indent + 2) + f"- Content of {file}:") + if file == "description.md": + result.append(" " * (indent + 4) + f"Please refer to the background of the scenario context.") + continue + with open(path, "r", encoding="utf-8") as f: + result.append(" " * (indent + 4) + f.read()) + if file_type == "tif": + result.append(" " * (indent + 2) + f"- Metadata of {file}:") + with Image.open(path) as img: + for tag, value in img.tag_v2.items(): + tag_name = TiffTags.TAGS_V2.get(tag, f"Unknown Tag {tag}") + result.append(" " * (indent + 4) + f"{tag_name}: {value}") + if file_type in ["json", "txt"]: + result.append(" " * (indent + 2) + f"- Content of {file}:") + with open(path, "r", encoding="utf-8") as f: + for i, line in enumerate(f): + if i < 2: + result.append( + " " * (indent + 4) + line.strip()[:100] + ("..." if len(line.strip()) > 100 else "") + ) + else: + break + + return "\n".join(result) + "\n" + + +""" data folder description version 2 """ +import json +from pathlib import Path + +import humanize +import pandas as pd +from genson import SchemaBuilder +from pandas.api.types import is_numeric_dtype + +# these files are treated as code (e.g. markdown wrapped) +code_files = {".py", ".sh", ".yaml", ".yml", ".md", ".html", ".xml", ".log", ".rst"} +# we treat these files as text (rather than binary) files +plaintext_files = {".txt", ".csv", ".json", ".tsv"} | code_files + + +def get_file_len_size(f: Path) -> tuple[int, str]: + """ + Calculate the size of a file (#lines for plaintext files, otherwise #bytes) + Also returns a human-readable string representation of the size. + """ + if f.suffix in plaintext_files: + num_lines = sum(1 for _ in open(f)) + return num_lines, f"{num_lines} lines" + else: + s = f.stat().st_size + return s, humanize.naturalsize(s) + + +def file_tree(path: Path, depth=0) -> str: + """Generate a tree structure of files in a directory""" + result = [] + files = [p for p in Path(path).iterdir() if not p.is_dir()] + dirs = [p for p in Path(path).iterdir() if p.is_dir()] + max_n = 4 if len(files) > 30 else 8 + for p in sorted(files)[:max_n]: + result.append(f"{' '*depth*4}{p.name} ({get_file_len_size(p)[1]})") + if len(files) > max_n: + result.append(f"{' '*depth*4}... and {len(files)-max_n} other files") + + for p in sorted(dirs): + result.append(f"{' '*depth*4}{p.name}/") + result.append(file_tree(p, depth + 1)) + + return "\n".join(result) + + +def _walk(path: Path): + """Recursively walk a directory (analogous to os.walk but for pathlib.Path)""" + for p in sorted(Path(path).iterdir()): + if p.is_dir(): + yield from _walk(p) + continue + yield p + + +def preview_csv(p: Path, file_name: str, simple=True) -> str: + """Generate a textual preview of a csv file + + Args: + p (Path): the path to the csv file + file_name (str): the file name to use in the preview + simple (bool, optional): whether to use a simplified version of the preview. Defaults to True. + + Returns: + str: the textual preview + """ + df = pd.read_csv(p) + + out = [] + + out.append(f"-> {file_name} has {df.shape[0]} rows and {df.shape[1]} columns.") + + if simple: + cols = df.columns.tolist() + sel_cols = 15 + cols_str = ", ".join(cols[:sel_cols]) + res = f"The columns are: {cols_str}" + if len(cols) > sel_cols: + res += f"... and {len(cols)-sel_cols} more columns" + out.append(res) + else: + out.append("Here is some information about the columns:") + for col in sorted(df.columns): + dtype = df[col].dtype + name = f"{col} ({dtype})" + + nan_count = df[col].isnull().sum() + + if dtype == "bool": + v = df[col][df[col].notnull()].mean() + out.append(f"{name} is {v*100:.2f}% True, {100-v*100:.2f}% False") + elif df[col].nunique() < 10: + out.append(f"{name} has {df[col].nunique()} unique values: {df[col].unique().tolist()}") + elif is_numeric_dtype(df[col]): + out.append(f"{name} has range: {df[col].min():.2f} - {df[col].max():.2f}, {nan_count} nan values") + elif dtype == "object": + out.append( + f"{name} has {df[col].nunique()} unique values. Some example values: {df[col].value_counts().head(4).index.tolist()}" + ) + + return "\n".join(out) + + +def preview_json(p: Path, file_name: str): + """Generate a textual preview of a json file using a generated json schema""" + builder = SchemaBuilder() + with open(p) as f: + first_line = f.readline().strip() + + try: + first_object = json.loads(first_line) + + if not isinstance(first_object, dict): + raise json.JSONDecodeError("The first line isn't JSON", first_line, 0) + + # if the the next line exists and is not empty, then it is a JSONL file + second_line = f.readline().strip() + if second_line: + f.seek(0) # so reset and read line by line + for line in f: + builder.add_object(json.loads(line.strip())) + # if it is empty, then it's a single JSON object file + else: + builder.add_object(first_object) + + except json.JSONDecodeError: + # if first line isn't JSON, then it's prettified and we can read whole file + f.seek(0) + builder.add_object(json.load(f)) + + return f"-> {file_name} has auto-generated json schema:\n" + builder.to_json(indent=2) + + +def describe_data_folder_v2(base_path, include_file_details=True, simple=False): + """ + Generate a textual preview of a directory, including an overview of the directory + structure and previews of individual files + """ + tree = f"```\n{file_tree(base_path)}```" + out = [tree] + + if include_file_details: + for fn in _walk(base_path): + file_name = str(fn.relative_to(base_path)) + + if fn.suffix == ".csv": + out.append(preview_csv(fn, file_name, simple=simple)) + elif fn.suffix == ".json": + out.append(preview_json(fn, file_name)) + elif fn.suffix in plaintext_files: + if get_file_len_size(fn)[0] < 30: + with open(fn) as f: + content = f.read() + if fn.suffix in code_files: + content = f"```\n{content}\n```" + out.append(f"-> {file_name} has content:\n\n{content}") + + result = "\n\n".join(out) + + # if the result is very long we generate a simpler version + if len(result) > 6_000 and not simple: + return describe_data_folder_v2(base_path, include_file_details=include_file_details, simple=True) + # if still too long, we truncate + if len(result) > 6_000 and simple: + return result[:6_000] + "\n... (truncated)" + + return result diff --git a/requirements.txt b/requirements.txt index 69720cf1..11f58e38 100644 --- a/requirements.txt +++ b/requirements.txt @@ -51,3 +51,7 @@ nbformat setuptools-scm seaborn azure.ai.inference + +# data folder desc +humanize +genson