2024-08-19 10:58:28 +08:00
import json
from pathlib import Path
2024-09-11 15:26:52 +08:00
import pandas as pd
2024-08-19 10:58:28 +08:00
from jinja2 import Environment , StrictUndefined
from rdagent.core.experiment import Experiment
from rdagent.core.prompts import Prompts
from rdagent.core.proposal import (
Hypothesis ,
HypothesisExperiment2Feedback ,
HypothesisFeedback ,
Trace ,
)
from rdagent.log import rdagent_logger as logger
from rdagent.oai.llm_utils import APIBackend
from rdagent.utils import convert2bool
2024-09-19 11:33:33 +08:00
prompt_dict = Prompts ( file_path = Path ( __file__ ) . parent . parent / "prompts.yaml" )
2024-08-19 10:58:28 +08:00
DIRNAME = Path ( __file__ ) . absolute () . resolve () . parent
2024-09-11 15:26:52 +08:00
class KGHypothesisExperiment2Feedback ( HypothesisExperiment2Feedback ):
2024-09-26 16:15:47 +08:00
def process_results ( self , current_result , sota_result ):
# Convert the results to dataframes
current_df = pd . DataFrame ( current_result )
sota_df = pd . DataFrame ( sota_result )
# Combine the dataframes on the Metric index
combined_df = pd . concat ([ current_df , sota_df ], axis = 1 )
combined_df . columns = [ "current_df" , "sota_df" ]
# combined_df["the largest"] = combined_df.apply(
# lambda row: "sota_df"
# if row["sota_df"] > row["current_df"]
# else ("Equal" if row["sota_df"] == row["current_df"] else "current_df"),
# axis=1,
# )
# Add a note about metric direction
evaluation_direction = "higher" if self . scen . evaluation_metric_direction else "lower"
2024-09-27 00:17:28 +08:00
evaluation_description = f "Direction of improvement (higher/lower is better) should be judged per metric. Here ' { evaluation_direction } ' is better for the metrics."
combined_df [ "Note" ] = evaluation_description
2024-09-26 16:15:47 +08:00
2024-09-27 00:17:28 +08:00
return combined_df , evaluation_description
2024-09-26 16:15:47 +08:00
2024-08-19 10:58:28 +08:00
def generate_feedback ( self , exp : Experiment , hypothesis : Hypothesis , trace : Trace ) -> HypothesisFeedback :
"""
The `ti` should be executed and the results should be included, as well as the comparison between previous results (done by LLM).
For example: `mlflow` of Qlib will be included.
"""
2024-09-11 15:26:52 +08:00
"""
Generate feedback for the given experiment and hypothesis.
Args:
exp: The experiment to generate feedback for.
hypothesis: The hypothesis to generate feedback for.
trace: The trace of the experiment.
Returns:
Any: The feedback generated for the given experiment and hypothesis.
"""
2024-08-19 10:58:28 +08:00
logger . info ( "Generating feedback..." )
2024-09-11 15:26:52 +08:00
hypothesis_text = hypothesis . hypothesis
current_result = exp . result
tasks_factors = []
if exp . sub_tasks :
tasks_factors = []
for task in exp . sub_tasks :
try :
task_info = task . get_task_information_and_implementation_result ()
tasks_factors . append ( task_info )
except AttributeError :
print ( f "Warning: Task { task } does not have get_task_information_and_implementation_result method" )
2024-08-19 10:58:28 +08:00
2024-09-27 00:17:28 +08:00
evaluation_description = None
2024-09-11 15:26:52 +08:00
# Check if there are any based experiments
if exp . based_experiments :
sota_result = exp . based_experiments [ - 1 ] . result
# Process the results to filter important metrics
2024-09-27 00:17:28 +08:00
combined_result , evaluation_description = self . process_results ( current_result , sota_result )
2024-09-11 15:26:52 +08:00
else :
# If there are no based experiments, we'll only use the current result
2024-09-27 00:17:28 +08:00
combined_result , evaluation_description = self . process_results (
current_result , current_result
) # Compare with itself
2024-09-11 15:26:52 +08:00
print ( "Warning: No previous experiments to compare against. Using current result as baseline." )
2024-08-19 10:58:28 +08:00
2024-09-20 22:01:21 +08:00
available_features = {
task_info : feature_shape for task_info , feature_shape in exp . experiment_workspace . data_description
}
model_code = exp . experiment_workspace . model_description
2024-09-20 16:06:49 +08:00
# Generate the user prompt based on the action type
if hypothesis . action == "Model tuning" :
prompt_key = "model_tuning_feedback_generation"
elif hypothesis . action == "Model feature selection" :
prompt_key = "feature_selection_feedback_generation"
else :
prompt_key = "factor_feedback_generation"
2024-09-11 15:26:52 +08:00
# Generate the system prompt
sys_prompt = (
2024-08-19 10:58:28 +08:00
Environment ( undefined = StrictUndefined )
2024-09-20 16:06:49 +08:00
. from_string ( prompt_dict [ prompt_key ][ "system" ])
2024-09-11 15:26:52 +08:00
. render ( scenario = self . scen . get_scenario_all_desc ())
)
2024-09-22 23:12:29 +08:00
last_task_and_code = None
if trace . hist :
last_task_and_code = (
trace . hist [ - 1 ][ 1 ] . experiment_workspace . data_description
if trace . hist [ - 1 ][ 0 ] . action == "Feature engineering" or trace . hist [ - 1 ][ 0 ] . action == "Feature processing"
else trace . hist [ - 1 ][ 1 ] . experiment_workspace . model_description
)
2024-09-20 16:06:49 +08:00
# Prepare render dictionary
render_dict = {
2024-09-24 17:23:18 +08:00
"last_hypothesis" : trace . hist [ - 1 ][ 0 ] if trace . hist else None ,
2024-09-22 23:12:29 +08:00
"last_task_and_code" : last_task_and_code ,
2024-09-20 16:06:49 +08:00
"last_result" : trace . hist [ - 1 ][ 1 ] . result if trace . hist else None ,
2024-09-28 00:40:25 +08:00
"sota_task_and_code" : (
exp . based_experiments [ - 1 ] . experiment_workspace . data_description if exp . based_experiments else None
),
2024-09-26 16:15:47 +08:00
"sota_result" : exp . based_experiments [ - 1 ] . result if exp . based_experiments else None ,
2024-09-20 16:06:49 +08:00
"hypothesis" : hypothesis ,
"exp" : exp ,
2024-09-26 16:15:47 +08:00
"model_code" : model_code , # This turn
"available_features" : available_features , # This turn
"combined_result" : combined_result , # This turn and sota
"hypothesis_text" : hypothesis_text , # This turn
"task_details" : tasks_factors , # This turn
2024-09-27 00:17:28 +08:00
"evaluation_description" : evaluation_description ,
2024-09-20 16:06:49 +08:00
}
2024-09-11 15:26:52 +08:00
usr_prompt = (
2024-09-20 16:06:49 +08:00
Environment ( undefined = StrictUndefined ) . from_string ( prompt_dict [ prompt_key ][ "user" ]) . render ( ** render_dict )
2024-08-19 10:58:28 +08:00
)
2024-09-11 15:26:52 +08:00
response = APIBackend () . build_messages_and_create_chat_completion (
user_prompt = usr_prompt ,
system_prompt = sys_prompt ,
2024-08-19 10:58:28 +08:00
json_mode = True ,
)
2024-09-11 15:26:52 +08:00
response_json = json . loads ( response )
observations = response_json . get ( "Observations" , "No observations provided" )
hypothesis_evaluation = response_json . get ( "Feedback for Hypothesis" , "No feedback provided" )
new_hypothesis = response_json . get ( "New Hypothesis" , "No new hypothesis provided" )
reason = response_json . get ( "Reasoning" , "No reasoning provided" )
decision = convert2bool ( response_json . get ( "Replace Best Result" , "no" ))
2024-09-30 04:15:07 +08:00
leaderboard = self . scen . leaderboard
current_score = current_result . iloc [ 0 ]
sorted_scores = sorted ( leaderboard , reverse = True )
import bisect
if self . scen . evaluation_metric_direction :
insert_position = bisect . bisect_right ([ - score for score in sorted_scores ], - current_score )
else :
insert_position = bisect . bisect_left ( sorted_scores , current_score , lo = 0 , hi = len ( sorted_scores ))
percentile_ranking = ( insert_position ) / ( len ( sorted_scores )) * 100
2024-09-11 15:26:52 +08:00
2024-09-19 11:33:33 +08:00
experiment_feedback = {
2024-09-28 00:40:25 +08:00
"current_competition" : self . scen . get_competition_full_desc (),
2024-09-19 11:33:33 +08:00
"hypothesis_text" : hypothesis_text ,
"current_result" : current_result ,
2024-09-27 00:17:28 +08:00
"model_code" : model_code ,
"available_features" : available_features ,
2024-09-19 11:33:33 +08:00
"observations" : observations ,
"hypothesis_evaluation" : hypothesis_evaluation ,
"reason" : reason ,
2024-09-30 04:15:07 +08:00
"percentile_ranking" : percentile_ranking ,
2024-09-19 11:33:33 +08:00
}
2024-09-27 00:17:28 +08:00
if self . scen . if_using_vector_rag :
self . scen . vector_base . add_experience_to_vector_base ( experiment_feedback )
2024-09-28 23:43:11 -04:00
self . scen . vector_base . dump ()
2024-09-27 00:17:28 +08:00
elif self . scen . if_using_graph_rag :
2024-09-28 00:40:25 +08:00
trace . knowledge_base . add_document ( experiment_feedback , self . scen )
2024-09-19 11:33:33 +08:00
2024-08-19 10:58:28 +08:00
return HypothesisFeedback (
2024-09-11 15:26:52 +08:00
observations = observations ,
hypothesis_evaluation = hypothesis_evaluation ,
new_hypothesis = new_hypothesis ,
reason = reason ,
decision = decision ,
2024-08-19 10:58:28 +08:00
)