From fd8a5d1c832d0c6773b5d082f268c80dc82ad9dd Mon Sep 17 00:00:00 2001 From: WinstonLiyt <104308117+WinstonLiyt@users.noreply.github.com> Date: Fri, 20 Sep 2024 22:01:21 +0800 Subject: [PATCH] feat: refine the code in model description and fix some bugs in feedback.py (#288) * fix some bugs in feedback.py * feat: kaggle templates related (#287) * add kaggle test * kaggle templates changes * rename two files * fix a grammar bug * fix a ci error * fix a bug --------- Co-authored-by: XianBW <36835909+XianBW@users.noreply.github.com> --- .../scenarios/kaggle/developer/feedback.py | 33 +----- rdagent/scenarios/kaggle/developer/runner.py | 104 ++++++++++++++---- .../{model_rf.py => model_randomforest.py} | 0 .../model/{model_xgb.py => model_xgboost.py} | 0 .../{model_rf.py => model_randomforest.py} | 0 .../model/{model_xgb.py => model_xgboost.py} | 0 .../scenarios/kaggle/experiment/prompts.yaml | 1 - .../scenarios/kaggle/experiment/scenario.py | 1 - .../scenarios/kaggle/experiment/workspace.py | 2 +- rdagent/scenarios/kaggle/prompts.yaml | 24 ++++ rdagent/utils/workflow.py | 5 + 11 files changed, 119 insertions(+), 51 deletions(-) rename rdagent/scenarios/kaggle/experiment/meta_tpl/model/{model_rf.py => model_randomforest.py} (100%) rename rdagent/scenarios/kaggle/experiment/meta_tpl/model/{model_xgb.py => model_xgboost.py} (100%) rename rdagent/scenarios/kaggle/experiment/playground-series-s4e8_template/model/{model_rf.py => model_randomforest.py} (100%) rename rdagent/scenarios/kaggle/experiment/playground-series-s4e8_template/model/{model_xgb.py => model_xgboost.py} (100%) diff --git a/rdagent/scenarios/kaggle/developer/feedback.py b/rdagent/scenarios/kaggle/developer/feedback.py index a32245f1..344316ea 100644 --- a/rdagent/scenarios/kaggle/developer/feedback.py +++ b/rdagent/scenarios/kaggle/developer/feedback.py @@ -46,32 +46,6 @@ def process_results(current_result, sota_result): class KGHypothesisExperiment2Feedback(HypothesisExperiment2Feedback): - def get_available_features(self, exp: Experiment): - features = [] - - for feature_info in exp.experiment_workspace.data_description: - task_info, feature_shape = feature_info - features.append( - {"name": task_info.factor_name, "description": task_info.factor_description, "shape": feature_shape} - ) - - return features - - def get_model_code(self, exp: Experiment): - model_type = exp.sub_tasks[0].model_type if exp.sub_tasks else None - if model_type == "XGBoost": - return exp.sub_workspace_list[0].code_dict.get( - "model_xgb.py" - ) # TODO Check if we need to replace this by using RepoAnalyzer - elif model_type == "RandomForest": - return exp.sub_workspace_list[0].code_dict.get("model_rf.py") - elif model_type == "LightGBM": - return exp.sub_workspace_list[0].code_dict.get("model_lgb.py") - elif model_type == "NN": - return exp.sub_workspace_list[0].code_dict.get("model_nn.py") - else: - return None - def generate_feedback(self, exp: Experiment, hypothesis: Hypothesis, trace: Trace) -> HypothesisFeedback: """ The `ti` should be executed and the results should be included, as well as the comparison between previous results (done by LLM). @@ -109,9 +83,10 @@ class KGHypothesisExperiment2Feedback(HypothesisExperiment2Feedback): combined_result = process_results(current_result, current_result) # Compare with itself print("Warning: No previous experiments to compare against. Using current result as baseline.") - available_features = self.get_available_features(exp) - # Get the appropriate model code - model_code = self.get_model_code(exp) + available_features = { + task_info: feature_shape for task_info, feature_shape in exp.experiment_workspace.data_description + } + model_code = exp.experiment_workspace.model_description # Generate the user prompt based on the action type if hypothesis.action == "Model tuning": diff --git a/rdagent/scenarios/kaggle/developer/runner.py b/rdagent/scenarios/kaggle/developer/runner.py index 64dcd388..5f6419df 100644 --- a/rdagent/scenarios/kaggle/developer/runner.py +++ b/rdagent/scenarios/kaggle/developer/runner.py @@ -1,20 +1,27 @@ +import json import pickle import shutil from pathlib import Path +from jinja2 import Environment, StrictUndefined + from rdagent.app.kaggle.conf import KAGGLE_IMPLEMENT_SETTING from rdagent.components.coder.factor_coder.config import FACTOR_IMPLEMENT_SETTINGS from rdagent.components.coder.factor_coder.factor import FactorTask +from rdagent.components.coder.model_coder.model import ModelTask from rdagent.components.runner import CachedRunner from rdagent.components.runner.conf import RUNNER_SETTINGS -from rdagent.core.exception import FactorEmptyError, ModelEmptyError +from rdagent.core.exception import CoderError, FactorEmptyError, ModelEmptyError from rdagent.core.experiment import ASpecificExp -from rdagent.oai.llm_utils import md5_hash +from rdagent.core.prompts import Prompts +from rdagent.oai.llm_utils import APIBackend, md5_hash from rdagent.scenarios.kaggle.experiment.kaggle_experiment import ( KGFactorExperiment, KGModelExperiment, ) +prompt_dict = Prompts(file_path=Path(__file__).parent.parent / "prompts.yaml") + class KGCachedRunner(CachedRunner[ASpecificExp]): def build_from_SOTA(self, exp: ASpecificExp) -> None: @@ -23,7 +30,7 @@ class KGCachedRunner(CachedRunner[ASpecificExp]): exp.experiment_workspace.data_description = exp.based_experiments[-1].experiment_workspace.data_description exp.experiment_workspace.model_description = exp.based_experiments[ -1 - ].experiment_workspace.model_description + ].experiment_workspace.model_description.copy() def get_cache_key(self, exp: ASpecificExp) -> str: codes = [] @@ -38,22 +45,19 @@ class KGCachedRunner(CachedRunner[ASpecificExp]): class KGModelRunner(KGCachedRunner[KGModelExperiment]): def develop(self, exp: KGModelExperiment) -> KGModelExperiment: self.build_from_SOTA(exp) - if exp.sub_workspace_list[0].target_task.model_type == "XGBoost": - if exp.sub_workspace_list[0].code_dict == {}: - raise ModelEmptyError("No model is implemented") - exp.experiment_workspace.inject_code(**{"model_xgb.py": exp.sub_workspace_list[0].code_dict["model.py"]}) - elif exp.sub_workspace_list[0].target_task.model_type == "RandomForest": - if exp.sub_workspace_list[0].code_dict == {}: - raise ModelEmptyError("No model is implemented") - exp.experiment_workspace.inject_code(**{"model_rf.py": exp.sub_workspace_list[0].code_dict["model.py"]}) - elif exp.sub_workspace_list[0].target_task.model_type == "LightGBM": - if exp.sub_workspace_list[0].code_dict == {}: - raise ModelEmptyError("No model is implemented") - exp.experiment_workspace.inject_code(**{"model_lgb.py": exp.sub_workspace_list[0].code_dict["model.py"]}) - elif exp.sub_workspace_list[0].target_task.model_type == "NN": - if exp.sub_workspace_list[0].code_dict == {}: - raise ModelEmptyError("No model is implemented") - exp.experiment_workspace.inject_code(**{"model_nn.py": exp.sub_workspace_list[0].code_dict["model.py"]}) + + sub_ws = exp.sub_workspace_list[0] + model_type = sub_ws.target_task.model_type + + if sub_ws.code_dict == {}: + raise ModelEmptyError("No model is implemented.") + else: + model_file_name = f"model_{model_type.lower()}.py" + exp.experiment_workspace.inject_code(**{model_file_name: sub_ws.code_dict["model.py"]}) + + model_description = sub_ws.target_task.get_task_information() + exp.experiment_workspace.model_description[model_type] = model_description + if RUNNER_SETTINGS.cache_result: cache_hit, result = self.get_cache_result(exp) if cache_hit: @@ -72,6 +76,48 @@ class KGModelRunner(KGCachedRunner[KGModelExperiment]): class KGFactorRunner(KGCachedRunner[KGFactorExperiment]): + def extract_model_task_from_code(self, code: str) -> str: + sys_prompt = ( + Environment(undefined=StrictUndefined) + .from_string(prompt_dict["extract_model_task_from_code"]["system"]) + .render() + ) + + user_prompt = ( + Environment(undefined=StrictUndefined) + .from_string(prompt_dict["extract_model_task_from_code"]["user"]) + .render(file_content=code) + ) + + model_task_description = APIBackend().build_messages_and_create_chat_completion( + user_prompt=user_prompt, + system_prompt=sys_prompt, + json_mode=True, + ) + + try: + response_json_analysis = json.loads(model_task_description) + task_desc = f"""name: {response_json_analysis['name']} + description: {response_json_analysis['description']} + """ + task_desc += ( + f"formulation: {response_json_analysis['formulation']}\n" + if response_json_analysis.get("formulation") + else "" + ) + task_desc += f"architecture: {response_json_analysis['architecture']}\n" + task_desc += ( + f"variables: {json.dumps(response_json_analysis['variables'], indent=4)}\n" + if response_json_analysis.get("variables") + else "" + ) + task_desc += f"hyperparameters: {json.dumps(response_json_analysis['hyperparameters'], indent=4)}\n" + task_desc += f"model_type: {response_json_analysis['model_type']}\n" + except json.JSONDecodeError: + task_desc = "Failed to parse LLM's response as JSON" + + return task_desc + def init_develop(self, exp: KGFactorExperiment) -> KGFactorExperiment: """ For the initial development, the experiment serves as a benchmark for feature engineering. @@ -100,6 +146,22 @@ class KGFactorRunner(KGCachedRunner[KGFactorExperiment]): feature_shape = org_data.shape[-1] exp.experiment_workspace.data_description.append((sub_task.get_task_information(), feature_shape)) + sub_model_1_description = ( + self.extract_model_task_from_code( + (exp.experiment_workspace.workspace_path / "model" / "model_randomforest.py").read_text() + ) + + f"""code: { (exp.experiment_workspace.workspace_path / "model" / "model_randomforest.py").read_text()}""" + ) + sub_model_2_description = ( + self.extract_model_task_from_code( + (exp.experiment_workspace.workspace_path / "model" / "model_xgboost.py").read_text() + ) + + f"""code: { (exp.experiment_workspace.workspace_path / "model" / "model_xgboost.py").read_text()}""" + ) + + exp.experiment_workspace.model_description["XGBoost"] = sub_model_1_description + exp.experiment_workspace.model_description["RandomForest"] = sub_model_2_description + if RUNNER_SETTINGS.cache_result: self.dump_cache_result(exp, result) @@ -133,7 +195,11 @@ class KGFactorRunner(KGCachedRunner[KGFactorExperiment]): result = exp.experiment_workspace.execute(run_env=env_to_use) + if result is None: + raise CoderError("No result is returned from the experiment workspace") + exp.result = result + if RUNNER_SETTINGS.cache_result: self.dump_cache_result(exp, result) diff --git a/rdagent/scenarios/kaggle/experiment/meta_tpl/model/model_rf.py b/rdagent/scenarios/kaggle/experiment/meta_tpl/model/model_randomforest.py similarity index 100% rename from rdagent/scenarios/kaggle/experiment/meta_tpl/model/model_rf.py rename to rdagent/scenarios/kaggle/experiment/meta_tpl/model/model_randomforest.py diff --git a/rdagent/scenarios/kaggle/experiment/meta_tpl/model/model_xgb.py b/rdagent/scenarios/kaggle/experiment/meta_tpl/model/model_xgboost.py similarity index 100% rename from rdagent/scenarios/kaggle/experiment/meta_tpl/model/model_xgb.py rename to rdagent/scenarios/kaggle/experiment/meta_tpl/model/model_xgboost.py diff --git a/rdagent/scenarios/kaggle/experiment/playground-series-s4e8_template/model/model_rf.py b/rdagent/scenarios/kaggle/experiment/playground-series-s4e8_template/model/model_randomforest.py similarity index 100% rename from rdagent/scenarios/kaggle/experiment/playground-series-s4e8_template/model/model_rf.py rename to rdagent/scenarios/kaggle/experiment/playground-series-s4e8_template/model/model_randomforest.py diff --git a/rdagent/scenarios/kaggle/experiment/playground-series-s4e8_template/model/model_xgb.py b/rdagent/scenarios/kaggle/experiment/playground-series-s4e8_template/model/model_xgboost.py similarity index 100% rename from rdagent/scenarios/kaggle/experiment/playground-series-s4e8_template/model/model_xgb.py rename to rdagent/scenarios/kaggle/experiment/playground-series-s4e8_template/model/model_xgboost.py diff --git a/rdagent/scenarios/kaggle/experiment/prompts.yaml b/rdagent/scenarios/kaggle/experiment/prompts.yaml index dbbe94db..06a4a4b0 100644 --- a/rdagent/scenarios/kaggle/experiment/prompts.yaml +++ b/rdagent/scenarios/kaggle/experiment/prompts.yaml @@ -8,7 +8,6 @@ kg_description_template: "Competition Type": "The type of competition, e.g., 'Classification', 'Regression', 'Clustering', 'Prediction", "Time-Series Forecasting", "Competition Description": "A brief description of the competition", "Target Description": "A description of the target variable to be predicted", - "Competition Features": "A dict of relevant features used in the competition and their descriptions (if available)", # if you are not sure about the meaning of the feature, please add a (guess) before the description. Importantly, your feature name should be exactly the same as the feature name in the dataset! } Since these might be very similar column names in data like one_hot_encoded columns, you can use some regex to group them together. diff --git a/rdagent/scenarios/kaggle/experiment/scenario.py b/rdagent/scenarios/kaggle/experiment/scenario.py index e3bb6bb3..c8221ea6 100644 --- a/rdagent/scenarios/kaggle/experiment/scenario.py +++ b/rdagent/scenarios/kaggle/experiment/scenario.py @@ -69,7 +69,6 @@ class KGScenario(Scenario): self.competition_type = response_json_analysis.get("Competition Type", "No type provided") self.competition_description = response_json_analysis.get("Competition Description", "No description provided") self.target_description = response_json_analysis.get("Target Description", "No target provided") - self.competition_features = response_json_analysis.get("Competition Features", "No features provided") self.competition_features = self.source_data @property diff --git a/rdagent/scenarios/kaggle/experiment/workspace.py b/rdagent/scenarios/kaggle/experiment/workspace.py index bd909f9a..7a0e9299 100644 --- a/rdagent/scenarios/kaggle/experiment/workspace.py +++ b/rdagent/scenarios/kaggle/experiment/workspace.py @@ -29,7 +29,7 @@ class KGFBWorkspace(FBWorkspace): super().__init__(*args, **kwargs) self.inject_code_from_folder(template_folder_path) self.data_description: list[str] = [] - self.model_description: str = "" + self.model_description: dict[str, str] = {} def generate_preprocess_data( self, diff --git a/rdagent/scenarios/kaggle/prompts.yaml b/rdagent/scenarios/kaggle/prompts.yaml index 00019b1e..36dbb353 100644 --- a/rdagent/scenarios/kaggle/prompts.yaml +++ b/rdagent/scenarios/kaggle/prompts.yaml @@ -263,3 +263,27 @@ feature_selection_feedback_generation: 4. Are there any domain-specific considerations that should inform our feature selection? Remember to focus on the select() method in the model code, as this is where feature selection is implemented. + +extract_model_task_from_code: + system: |- + You are an expert in analyzing code for machine learning models. + user: |- + Given the following code, summarize the machine learning model including: + - Model architecture + - Hyperparameters + - Formulation and variables + - Model type (one of XGBoost, RandomForest, LightGBM, NN) + + Code: + {{ file_content }} + + Return the information in JSON format with the following structure: + { + "name": "", + "description": "", + "architecture": "", + "hyperparameters": {}, + "formulation": "", + "variables": {}, + "model_type": "" + } diff --git a/rdagent/utils/workflow.py b/rdagent/utils/workflow.py index c9c6f1c0..b49b6af6 100644 --- a/rdagent/utils/workflow.py +++ b/rdagent/utils/workflow.py @@ -17,6 +17,7 @@ from typing import Callable from tqdm.auto import tqdm +from rdagent.core.exception import CoderError from rdagent.log import rdagent_logger as logger @@ -114,6 +115,10 @@ class LoopBase: self.loop_idx += 1 self.step_idx = 0 continue + except CoderError as e: + logger.warning(f"Traceback loop {li} due to {e}") + self.step_idx -= 1 + continue end = datetime.datetime.now(datetime.timezone.utc)