2025-04-09 23:24:12 +08:00
from typing import Literal
2025-03-05 18:30:02 +08:00
from pydantic_settings import SettingsConfigDict
2025-01-17 22:53:05 +08:00
from rdagent.app.kaggle.conf import KaggleBasePropSetting
class DataScienceBasePropSetting ( KaggleBasePropSetting ):
2025-05-06 16:00:13 +08:00
# TODO: Kaggle Setting should be the subclass of DataScience
2025-03-05 18:30:02 +08:00
model_config = SettingsConfigDict ( env_prefix = "DS_" , protected_namespaces = ())
2025-01-17 22:53:05 +08:00
# Main components
## Scen
scen : str = "rdagent.scenarios.data_science.scen.KaggleScen"
2025-06-17 14:55:11 +08:00
"""
Scenario class for data science tasks.
- For Kaggle competitions, use: "rdagent.scenarios.data_science.scen.KaggleScen"
- For custom data science scenarios, use: "rdagent.scenarios.data_science.scen.DataScienceScen"
"""
2025-01-17 22:53:05 +08:00
2025-05-28 11:14:37 +08:00
hypothesis_gen : str = "rdagent.scenarios.data_science.proposal.exp_gen.proposal.DSProposalV2ExpGen"
2025-05-09 15:38:25 +08:00
"""Hypothesis generation class"""
2025-07-09 18:36:42 +08:00
summarizer : str = "rdagent.scenarios.data_science.dev.feedback.DSExperiment2Feedback"
summarizer_init_kwargs : dict = {
"version" : "exp_feedback" ,
}
2025-01-24 15:03:21 +08:00
## Workflow Related
2025-01-24 13:34:47 +08:00
consecutive_errors : int = 5
2025-04-30 17:36:35 +08:00
## Coding Related
coding_fail_reanalyze_threshold : int = 3
2025-01-24 15:03:21 +08:00
debug_timeout : int = 600
"""The timeout limit for running on debugging data"""
full_timeout : int = 3600
"""The timeout limit for running on full data"""
2025-03-27 17:17:41 +08:00
### specific feature
#### enable specification
spec_enabled : bool = True
2025-05-06 16:00:13 +08:00
#### proposal related
2025-07-08 15:22:39 +08:00
# proposal_version: str = "v2" deprecated
coder_on_whole_pipeline : bool = True
2025-04-07 13:01:29 +08:00
max_trace_hist : int = 3
2025-04-02 13:37:47 +08:00
2025-04-02 00:18:42 -06:00
coder_max_loop : int = 10
2025-07-10 18:10:32 +08:00
runner_max_loop : int = 3
2025-04-02 00:18:42 -06:00
2025-07-10 18:10:32 +08:00
sample_data_by_LLM : bool = True
2025-05-06 16:00:13 +08:00
use_raw_description : bool = False
2025-05-10 17:02:12 +08:00
show_nan_columns : bool = False
2025-04-08 03:27:21 -06:00
2025-05-06 16:00:13 +08:00
#### model dump
2025-04-09 23:24:12 +08:00
enable_model_dump : bool = False
2025-04-10 20:12:21 +08:00
enable_doc_dev : bool = False
2025-04-09 23:24:12 +08:00
model_dump_check_level : Literal [ "medium" , "high" ] = "medium"
2025-04-18 14:01:03 +08:00
### knowledge base
enable_knowledge_base : bool = False
knowledge_base_version : str = "v1"
knowledge_base_path : str | None = None
idea_pool_json_path : str | None = None
### archive log folder after each loop
enable_log_archive : bool = True
log_archive_path : str | None = None
log_archive_temp_path : str | None = (
None # This is to store the mid tar file since writing the tar file is preferred in local storage then copy to target storage
)
2025-05-06 16:00:13 +08:00
#### Evaluation on Test related
eval_sub_dir : str = "eval" # TODO: fixme, this is not a good name
"""We'll use f" {DS_RD_SETTING.local_data_path} / {DS_RD_SETTING.eval_sub_dir} / {competition} "
to find the scriipt to evaluate the submission on test"""
2025-05-19 17:59:42 +08:00
"""---below are the settings for multi-trace---"""
### multi-trace related
max_trace_num : int = 3
"""The maximum number of traces to grow before merging"""
#### multi-trace:checkpoint selector
2025-07-11 17:42:40 +08:00
selector_name : str = "rdagent.scenarios.data_science.proposal.exp_gen.select.expand.LatestCKPSelector"
2025-05-19 17:59:42 +08:00
"""The name of the selector to use"""
sota_count_window : int = 5
"""The number of trials to consider for SOTA count"""
sota_count_threshold : int = 1
"""The threshold for SOTA count"""
#### multi-trace: SOTA experiment selector
2025-07-11 17:42:40 +08:00
sota_exp_selector_name : str = "rdagent.scenarios.data_science.proposal.exp_gen.select.submit.GlobalSOTASelector"
2025-05-19 17:59:42 +08:00
"""The name of the SOTA experiment selector to use"""
### multi-trace:inject optimals for multi-trace
# inject diverse when start a new sub-trace
2025-05-09 15:38:25 +08:00
enable_inject_diverse : bool = False
2025-06-11 23:13:30 +08:00
# inject knowledge at the root of the trace
2025-05-19 17:59:42 +08:00
enable_inject_knowledge_at_root : bool = False
2025-06-11 23:13:30 +08:00
# enable different version of DSExpGen for multi-trace
enable_multi_version_exp_gen : bool = False
exp_gen_version_list : str = "v3,v2"
2025-05-19 17:59:42 +08:00
#### multi-trace: time for final multi-trace merge
merge_hours : int = 2
"""The time for merge"""
2025-05-22 16:10:00 +08:00
#### multi-trace: max SOTA-retrieved number, used in AutoSOTAexpSelector
# constrains the number of SOTA experiments to retrieve, otherwise too many SOTA experiments to retrieve will cause the exceed of the context window of LLM
max_sota_retrieved_num : int = 10
"""The maximum number of SOTA experiments to retrieve in a LLM call"""
2025-01-17 22:53:05 +08:00
DS_RD_SETTING = DataScienceBasePropSetting ()