2025-04-03 13:00:11 +08:00
# tess successfully running.
# (GPT) if it aligns with the spec & rationality of the spec.
import json
import re
from pathlib import Path
import pandas as pd
from rdagent.app.data_science.conf import DS_RD_SETTING
from rdagent.components.coder.CoSTEER import CoSTEERMultiFeedback
from rdagent.components.coder.CoSTEER.evaluators import (
CoSTEEREvaluator ,
CoSTEERSingleFeedback ,
)
from rdagent.components.coder.CoSTEER.knowledge_management import (
CoSTEERQueriedKnowledgeV2 ,
)
2025-04-09 23:24:12 +08:00
from rdagent.components.coder.data_science.conf import get_clear_ws_cmd , get_ds_env
2025-04-17 16:19:26 +08:00
from rdagent.components.coder.data_science.utils import remove_eda_part
2025-04-03 13:00:11 +08:00
from rdagent.core.experiment import FBWorkspace , Task
2025-05-06 16:00:13 +08:00
from rdagent.scenarios.data_science.test_eval import get_test_eval
2025-04-03 13:00:11 +08:00
from rdagent.utils.agent.tpl import T
from rdagent.utils.agent.workflow import build_cls_from_json_with_retry
DIRNAME = Path ( __file__ ) . absolute () . resolve () . parent
PipelineSingleFeedback = CoSTEERSingleFeedback
PipelineMultiFeedback = CoSTEERMultiFeedback
class PipelineCoSTEEREvaluator ( CoSTEEREvaluator ):
def evaluate (
self ,
target_task : Task ,
implementation : FBWorkspace ,
gt_implementation : FBWorkspace ,
queried_knowledge : CoSTEERQueriedKnowledgeV2 = None ,
** kwargs ,
) -> PipelineSingleFeedback :
target_task_information = target_task . get_task_information ()
if (
queried_knowledge is not None
and target_task_information in queried_knowledge . success_task_to_knowledge_dict
):
return queried_knowledge . success_task_to_knowledge_dict [ target_task_information ] . feedback
elif queried_knowledge is not None and target_task_information in queried_knowledge . failed_task_info_set :
return PipelineSingleFeedback (
execution = "This task has failed too many times, skip implementation." ,
return_checking = "This task has failed too many times, skip implementation." ,
code = "This task has failed too many times, skip implementation." ,
final_decision = False ,
)
2025-05-06 16:00:13 +08:00
env = get_ds_env ( extra_volumes = { self . scen . debug_path : T ( "scenarios.data_science.share:scen.input_path" ) . r ()})
2025-04-03 13:00:11 +08:00
# Clean the scores.csv & submission.csv.
2025-04-09 23:24:12 +08:00
implementation . execute ( env = env , entry = get_clear_ws_cmd ())
2025-05-06 16:00:13 +08:00
stdout , execute_ret_code = implementation . execute_ret_code ( env = env , entry = f "python -m coverage run main.py" )
2025-04-17 16:19:26 +08:00
stdout = remove_eda_part ( stdout )
2025-05-06 16:00:13 +08:00
stdout += f "The code executed { 'successfully' if execute_ret_code == 0 else 'failed' } ."
2025-04-03 13:00:11 +08:00
score_fp = implementation . workspace_path / "scores.csv"
score_ret_code = 0
score_check_text = ""
if not score_fp . exists ():
score_check_text = "[Error] Metrics file (scores.csv) is not generated!"
score_ret_code = 1
else :
try :
score_df = pd . read_csv ( score_fp , index_col = 0 )
model_set_in_scores = set ( score_df . index )
# Check model names (index)
2025-04-07 19:06:42 +08:00
if not score_df . index . is_unique :
2025-06-23 10:38:27 +08:00
score_check_text += " \n [Error] The file 'scores.csv' contains duplicate model names."
2025-04-07 19:06:42 +08:00
score_ret_code = 1
2025-04-03 13:00:11 +08:00
if "ensemble" not in model_set_in_scores :
2025-06-23 10:38:27 +08:00
score_check_text += " \n [Error] The file 'scores.csv' doesn't contain the ensemble model."
2025-04-03 13:00:11 +08:00
score_ret_code = 1
2025-04-07 19:06:42 +08:00
if score_ret_code != 0 :
2025-06-23 10:38:27 +08:00
score_check_text += f "The dataframe in file 'scores.csv' is: \n { score_df } "
2025-04-03 13:00:11 +08:00
# Check metric name (columns)
if score_df . columns . tolist () != [ self . scen . metric_name ]:
score_check_text += f " \n [Error] The scores dataframe does not contain the correct column names. \n Correct columns is: [' { self . scen . metric_name } '] \n But got: { score_df . columns . tolist () } "
score_ret_code = 1
2025-04-04 22:44:55 +08:00
# Check if scores contain NaN (values)
if score_df . isnull () . values . any ():
nan_locations = score_df [ score_df . isnull () . any ( axis = 1 )]
score_check_text += f " \n [Error] The scores dataframe contains NaN values at the following locations: \n { nan_locations } "
score_ret_code = 1
2025-04-03 13:00:11 +08:00
except Exception as e :
score_check_text += f " \n [Error] in checking the scores.csv file: { e } \n scores.csv's content: \n ----- \n { score_fp . read_text () } \n -----"
score_ret_code = 1
2025-05-06 16:00:13 +08:00
test_eval = get_test_eval ()
if not test_eval . is_sub_enabled ( self . scen . competition ):
submission_ret_code = 0
else :
# Check submission file
base_check_code = T ( ".eval_tests.submission_format_test" , ftype = "txt" ) . r ()
implementation . inject_files ( ** { "test/submission_format_test.py" : base_check_code })
# stdout += "----Submission Check 1-----\n"
submission_check_out , submission_ret_code = implementation . execute_ret_code (
env = env , entry = "python test/submission_format_test.py"
)
if DS_RD_SETTING . rule_base_eval :
if execute_ret_code == 0 and score_ret_code == 0 and submission_ret_code == 0 :
return PipelineSingleFeedback (
execution = stdout ,
return_checking = score_check_text + " \n " + submission_check_out ,
code = "Code evaluation is not available." ,
final_decision = True ,
)
else :
return PipelineSingleFeedback (
execution = stdout ,
return_checking = score_check_text + " \n " + submission_check_out ,
code = "Code evaluation is not available." ,
final_decision = False ,
)
stdout += " \n " + submission_check_out
2025-04-03 13:00:11 +08:00
2025-05-09 15:38:25 +08:00
if not isinstance ( implementation , FBWorkspace ):
eda_output = None
else :
eda_output = implementation . file_dict . get ( "EDA.md" , None )
2025-04-03 13:00:11 +08:00
system_prompt = T ( ".prompts:pipeline_eval.system" ) . r (
2025-04-09 09:42:30 +08:00
scenario = self . scen . get_scenario_all_desc ( eda_output = eda_output ),
2025-04-03 13:00:11 +08:00
task_desc = target_task . get_task_information (),
2025-05-06 16:00:13 +08:00
is_sub_enabled = test_eval . is_sub_enabled ( self . scen . competition ),
2025-04-03 13:00:11 +08:00
spec = T ( "scenarios.data_science.share:component_spec.Pipeline" ) . r (),
)
user_prompt = T ( ".prompts:pipeline_eval.user" ) . r (
stdout = stdout . strip (),
code = implementation . file_dict [ "main.py" ],
)
wfb = build_cls_from_json_with_retry (
PipelineSingleFeedback ,
system_prompt = system_prompt ,
user_prompt = user_prompt ,
init_kwargs_update_func = PipelineSingleFeedback . val_and_update_init_dict ,
)
if score_ret_code != 0 :
wfb . final_decision = False
wfb . return_checking += " \n " + score_check_text
if submission_ret_code != 0 :
wfb . final_decision = False
wfb . return_checking += " \n Submission file check failed."
return wfb