Files
NexQuant/rdagent/app/data_science/conf.py
T

120 lines
4.1 KiB
Python
Raw Normal View History

2025-04-09 23:24:12 +08:00
from typing import Literal
from pydantic_settings import SettingsConfigDict
from rdagent.app.kaggle.conf import KaggleBasePropSetting
class DataScienceBasePropSetting(KaggleBasePropSetting):
2025-05-06 16:00:13 +08:00
# TODO: Kaggle Setting should be the subclass of DataScience
model_config = SettingsConfigDict(env_prefix="DS_", protected_namespaces=())
# Main components
## Scen
scen: str = "rdagent.scenarios.data_science.scen.KaggleScen"
"""
Scenario class for data science tasks.
- For Kaggle competitions, use: "rdagent.scenarios.data_science.scen.KaggleScen"
- For custom data science scenarios, use: "rdagent.scenarios.data_science.scen.DataScienceScen"
"""
2025-05-28 11:14:37 +08:00
hypothesis_gen: str = "rdagent.scenarios.data_science.proposal.exp_gen.proposal.DSProposalV2ExpGen"
2025-05-09 15:38:25 +08:00
"""Hypothesis generation class"""
2025-07-09 18:36:42 +08:00
summarizer: str = "rdagent.scenarios.data_science.dev.feedback.DSExperiment2Feedback"
summarizer_init_kwargs: dict = {
"version": "exp_feedback",
}
## Workflow Related
2025-01-24 13:34:47 +08:00
consecutive_errors: int = 5
## Coding Related
coding_fail_reanalyze_threshold: int = 3
debug_timeout: int = 600
"""The timeout limit for running on debugging data"""
full_timeout: int = 3600
"""The timeout limit for running on full data"""
2025-03-27 17:17:41 +08:00
### specific feature
#### enable specification
spec_enabled: bool = True
2025-05-06 16:00:13 +08:00
#### proposal related
# proposal_version: str = "v2" deprecated
coder_on_whole_pipeline: bool = True
max_trace_hist: int = 3
2025-04-02 13:37:47 +08:00
coder_max_loop: int = 10
runner_max_loop: int = 3
sample_data_by_LLM: bool = True
2025-05-06 16:00:13 +08:00
use_raw_description: bool = False
2025-05-10 17:02:12 +08:00
show_nan_columns: bool = False
2025-05-06 16:00:13 +08:00
#### model dump
2025-04-09 23:24:12 +08:00
enable_model_dump: bool = False
enable_doc_dev: bool = False
2025-04-09 23:24:12 +08:00
model_dump_check_level: Literal["medium", "high"] = "medium"
### knowledge base
enable_knowledge_base: bool = False
knowledge_base_version: str = "v1"
knowledge_base_path: str | None = None
idea_pool_json_path: str | None = None
### archive log folder after each loop
enable_log_archive: bool = True
log_archive_path: str | None = None
log_archive_temp_path: str | None = (
None # This is to store the mid tar file since writing the tar file is preferred in local storage then copy to target storage
)
2025-05-06 16:00:13 +08:00
#### Evaluation on Test related
eval_sub_dir: str = "eval" # TODO: fixme, this is not a good name
"""We'll use f"{DS_RD_SETTING.local_data_path}/{DS_RD_SETTING.eval_sub_dir}/{competition}"
to find the scriipt to evaluate the submission on test"""
2025-05-19 17:59:42 +08:00
"""---below are the settings for multi-trace---"""
### multi-trace related
max_trace_num: int = 3
"""The maximum number of traces to grow before merging"""
#### multi-trace:checkpoint selector
selector_name: str = "rdagent.scenarios.data_science.proposal.exp_gen.select.expand.LatestCKPSelector"
2025-05-19 17:59:42 +08:00
"""The name of the selector to use"""
sota_count_window: int = 5
"""The number of trials to consider for SOTA count"""
sota_count_threshold: int = 1
"""The threshold for SOTA count"""
#### multi-trace: SOTA experiment selector
sota_exp_selector_name: str = "rdagent.scenarios.data_science.proposal.exp_gen.select.submit.GlobalSOTASelector"
2025-05-19 17:59:42 +08:00
"""The name of the SOTA experiment selector to use"""
### multi-trace:inject optimals for multi-trace
# inject diverse when start a new sub-trace
2025-05-09 15:38:25 +08:00
enable_inject_diverse: bool = False
# inject knowledge at the root of the trace
2025-05-19 17:59:42 +08:00
enable_inject_knowledge_at_root: bool = False
# enable different version of DSExpGen for multi-trace
enable_multi_version_exp_gen: bool = False
exp_gen_version_list: str = "v3,v2"
2025-05-19 17:59:42 +08:00
#### multi-trace: time for final multi-trace merge
merge_hours: int = 2
"""The time for merge"""
#### multi-trace: max SOTA-retrieved number, used in AutoSOTAexpSelector
# constrains the number of SOTA experiments to retrieve, otherwise too many SOTA experiments to retrieve will cause the exceed of the context window of LLM
max_sota_retrieved_num: int = 10
"""The maximum number of SOTA experiments to retrieve in a LLM call"""
DS_RD_SETTING = DataScienceBasePropSetting()