mirror of
https://github.com/NicolasBohn/NexQuant.git
synced 2026-07-29 08:27:43 +00:00
eb5f69024a
* chore: print up to 100 columns in simple mode * fix: check content for model dump * chore: add show_nan_columns config
91 lines
2.9 KiB
Python
91 lines
2.9 KiB
Python
from typing import Literal
|
|
|
|
from pydantic_settings import SettingsConfigDict
|
|
|
|
from rdagent.app.kaggle.conf import KaggleBasePropSetting
|
|
|
|
|
|
class DataScienceBasePropSetting(KaggleBasePropSetting):
|
|
# TODO: Kaggle Setting should be the subclass of DataScience
|
|
model_config = SettingsConfigDict(env_prefix="DS_", protected_namespaces=())
|
|
|
|
# Main components
|
|
## Scen
|
|
scen: str = "rdagent.scenarios.data_science.scen.KaggleScen"
|
|
"""Scenario class for data mining model"""
|
|
|
|
hypothesis_gen: str = "rdagent.scenarios.data_science.proposal.exp_gen.DSExpGen"
|
|
"""Hypothesis generation class"""
|
|
|
|
## Workflow Related
|
|
consecutive_errors: int = 5
|
|
|
|
## Coding Related
|
|
coding_fail_reanalyze_threshold: int = 3
|
|
|
|
debug_timeout: int = 600
|
|
"""The timeout limit for running on debugging data"""
|
|
full_timeout: int = 3600
|
|
"""The timeout limit for running on full data"""
|
|
|
|
### specific feature
|
|
|
|
#### enable specification
|
|
spec_enabled: bool = True
|
|
|
|
#### proposal related
|
|
proposal_version: str = "v1"
|
|
coder_on_whole_pipeline: bool = False
|
|
max_trace_hist: int = 3
|
|
|
|
coder_max_loop: int = 10
|
|
runner_max_loop: int = 1
|
|
|
|
rule_base_eval: bool = False
|
|
sample_data: bool = True
|
|
use_raw_description: bool = False
|
|
show_nan_columns: bool = False
|
|
|
|
#### model dump
|
|
enable_model_dump: bool = False
|
|
enable_doc_dev: bool = False
|
|
model_dump_check_level: Literal["medium", "high"] = "medium"
|
|
|
|
### selector related
|
|
|
|
#### checkpoint selector related
|
|
# selector_name: str = "latest"
|
|
selector_name: str = "rdagent.scenarios.data_science.proposal.exp_gen.ckp_select.LatestCKPSelector"
|
|
"""The name of the selector to use"""
|
|
sota_count_window: int = 5
|
|
"""The number of trials to consider for SOTA count"""
|
|
sota_count_threshold: int = 1
|
|
"""The threshold for SOTA count"""
|
|
|
|
#### SOTA experiment selector related
|
|
sota_exp_selector_name: str = "rdagent.scenarios.data_science.proposal.exp_gen.sota_exp_select.GlobalSOTASelector"
|
|
"""The name of the SOTA experiment selector to use"""
|
|
### knowledge base
|
|
enable_knowledge_base: bool = False
|
|
knowledge_base_version: str = "v1"
|
|
knowledge_base_path: str | None = None
|
|
idea_pool_json_path: str | None = None
|
|
|
|
### archive log folder after each loop
|
|
enable_log_archive: bool = True
|
|
log_archive_path: str | None = None
|
|
log_archive_temp_path: str | None = (
|
|
None # This is to store the mid tar file since writing the tar file is preferred in local storage then copy to target storage
|
|
)
|
|
|
|
#### Evaluation on Test related
|
|
eval_sub_dir: str = "eval" # TODO: fixme, this is not a good name
|
|
"""We'll use f"{DS_RD_SETTING.local_data_path}/{DS_RD_SETTING.eval_sub_dir}/{competition}"
|
|
to find the scriipt to evaluate the submission on test"""
|
|
|
|
### inject diverse
|
|
enable_inject_diverse: bool = False
|
|
|
|
|
|
DS_RD_SETTING = DataScienceBasePropSetting()
|