reporeformat V2 (#23)

* reformat factor implement process

* move some code to more reasonable place

* fix the bug

* add test function in factor_extract_and_implement.py

* change select factor number to ratio , add some factor implement setting and fix some bug while using knowledgebase

* change evoagent

* add abstract class EvoAgent

* add benchmark workflow

* fix some bug in llm_utils

* run wenjun's code

* fix the knowledgebase instance check

---------

Co-authored-by: xuyang1 <xuyang1@microsoft.com>
This commit is contained in:
USTCKevinF
2024-06-14 12:59:44 +08:00
committed by GitHub
parent 9e82da243b
commit ebb659a018
32 changed files with 2227 additions and 1141 deletions
+27
View File
@@ -0,0 +1,27 @@
from dotenv import load_dotenv
load_dotenv(verbose=True, override=True)
from dataclasses import field
from pathlib import Path
from typing import Literal, Optional, Union
from pydantic_settings import BaseSettings
DIRNAME = Path(__file__).absolute().resolve().parent
BENCHMARK_VERSION = Literal["paper", "amcV01", "amcV02train", "amcV02test"]
class BenchmarkSettings(BaseSettings):
ground_truth_dir: Path = DIRNAME / "ground_truth"
bench_version: Union[BENCHMARK_VERSION, str] = "paper"
bench_test_round: int = 20
bench_test_case_n: Optional[int] = None # how many test cases to run; If not given, all test cases will be run
bench_method_cls: str = "scripts.factor_implementation.baselines.naive.one_shot.OneshotFactorGen"
bench_method_extra_kwargs: dict = field(
default_factory=dict,
) # extra kwargs for the method to be tested except the task list
bench_result_path: Path = DIRNAME / "result"