update code (#9)

Co-authored-by: xuyang1 <xuyang1@microsoft.com>
This commit is contained in:
Xu Yang
2024-05-21 22:48:41 +08:00
committed by GitHub
parent 230dca2bc2
commit c6833b0858
27 changed files with 5564 additions and 0 deletions
@@ -0,0 +1,231 @@
from __future__ import annotations
import re
from typing import List
from pandas.core.api import DataFrame as DataFrame
from core.evolving_framework import Evaluator as EvolvingEvaluator
from core.evolving_framework import Feedback, QueriedKnowledge
from core.log import FinCoLog
from factor_implementation.evolving.evolvable_subjects import (
FactorImplementationList,
)
from factor_implementation.share_modules.conf import FactorImplementSettings
from factor_implementation.share_modules.evaluator import (
Evaluator as FactorImplementationEvaluator,
)
from factor_implementation.share_modules.evaluator import (
FactorImplementationCodeEvaluator,
FactorImplementationFinalDecisionEvaluator,
FactorImplementationValueEvaluator,
)
from factor_implementation.share_modules.factor import (
FactorImplementation,
FactorImplementationTask,
)
from core.utils import multiprocessing_wrapper
class FactorImplementationSingleFeedback:
"""This class is a feedback to single implementation which is generated from an evaluator."""
def __init__(
self,
execution_feedback: str = None,
value_generated_flag: bool = False,
code_feedback: str = None,
factor_value_feedback: str = None,
final_decision: bool = None,
final_feedback: str = None,
final_decision_based_on_gt: bool = None,
) -> None:
self.execution_feedback = execution_feedback
self.value_generated_flag = value_generated_flag
self.code_feedback = code_feedback
self.factor_value_feedback = factor_value_feedback
self.final_decision = final_decision
self.final_feedback = final_feedback
self.final_decision_based_on_gt = final_decision_based_on_gt
def __str__(self) -> str:
return f"""------------------Factor Execution Feedback------------------
{self.execution_feedback}
------------------Factor Code Feedback------------------
{self.code_feedback}
------------------Factor Value Feedback------------------
{self.factor_value_feedback}
------------------Factor Final Feedback------------------
{self.final_feedback}
------------------Factor Final Decision------------------
This implementation is {'SUCCESS' if self.final_decision else 'FAIL'}.
"""
class FactorImplementationsMultiFeedback(
Feedback,
List[FactorImplementationSingleFeedback],
):
"""Feedback contains a list, each element is the corresponding feedback for each factor implementation."""
class FactorImplementationEvaluatorV1(FactorImplementationEvaluator):
"""This class is the v1 version of evaluator for a single factor implementation.
It calls several evaluators in share modules to evaluate the factor implementation.
"""
def __init__(self) -> None:
self.code_evaluator = FactorImplementationCodeEvaluator()
self.value_evaluator = FactorImplementationValueEvaluator()
self.final_decision_evaluator = FactorImplementationFinalDecisionEvaluator()
def evaluate(
self,
target_task: FactorImplementationTask,
implementation: FactorImplementation,
gt_implementation: FactorImplementation = None,
queried_knowledge: QueriedKnowledge = None,
**kwargs,
) -> FactorImplementationSingleFeedback:
if implementation is None:
return None
target_task_information = target_task.get_factor_information()
if (
queried_knowledge is not None
and target_task_information in queried_knowledge.success_task_to_knowledge_dict
):
return queried_knowledge.success_task_to_knowledge_dict[target_task_information].feedback
elif queried_knowledge is not None and target_task_information in queried_knowledge.failed_task_info_set:
return FactorImplementationSingleFeedback(
execution_feedback="This task has failed too many times, skip implementation.",
value_generated_flag=False,
code_feedback="This task has failed too many times, skip code evaluation.",
factor_value_feedback="This task has failed too many times, skip value evaluation.",
final_decision=False,
final_feedback="This task has failed too many times, skip final decision evaluation.",
final_decision_based_on_gt=False,
)
else:
factor_feedback = FactorImplementationSingleFeedback()
(
factor_feedback.execution_feedback,
source_df,
) = implementation.execute()
# Remove the long list of numbers in the feedback
pattern = r"(?<=\D)(,\s+-?\d+\.\d+){50,}(?=\D)"
factor_feedback.execution_feedback = re.sub(pattern, ", ", factor_feedback.execution_feedback)
execution_feedback_lines = [
line for line in factor_feedback.execution_feedback.split("\n") if "warning" not in line.lower()
]
factor_feedback.execution_feedback = "\n".join(execution_feedback_lines)
if source_df is None:
factor_feedback.factor_value_feedback = "No factor value generated, skip value evaluation."
factor_feedback.value_generated_flag = False
value_decision = None
else:
factor_feedback.value_generated_flag = True
if gt_implementation is not None:
_, gt_df = gt_implementation.execute(store_result=True)
else:
gt_df = None
try:
source_df = source_df.sort_index()
if gt_df is not None:
gt_df = gt_df.sort_index()
(
factor_feedback.factor_value_feedback,
value_decision,
) = self.value_evaluator.evaluate(source_df=source_df, gt_df=gt_df)
except Exception as e:
FinCoLog().warning("Value evaluation failed with exception: %s", e)
factor_feedback.factor_value_feedback = "Value evaluation failed."
value_decision = False
factor_feedback.final_decision_based_on_gt = gt_implementation is not None
if value_decision is not None and value_decision is True:
# To avoid confusion, when value_decision is True, we do not need code feedback
factor_feedback.code_feedback = "Final decision is True and there are no code critics."
factor_feedback.final_decision = value_decision
factor_feedback.final_feedback = "Value evaluation passed, skip final decision evaluation."
else:
factor_feedback.code_feedback = self.code_evaluator.evaluate(
target_task=target_task,
implementation=implementation,
execution_feedback=factor_feedback.execution_feedback,
value_feedback=factor_feedback.factor_value_feedback,
gt_implementation=gt_implementation,
)
(
factor_feedback.final_decision,
factor_feedback.final_feedback,
) = self.final_decision_evaluator.evaluate(
target_task=target_task,
execution_feedback=factor_feedback.execution_feedback,
value_feedback=factor_feedback.factor_value_feedback,
code_feedback=factor_feedback.code_feedback,
)
return factor_feedback
class FactorImplementationsMultiEvaluator(EvolvingEvaluator):
def __init__(self, single_evaluator=FactorImplementationEvaluatorV1()) -> None:
super().__init__()
self.single_factor_implementation_evaluator = single_evaluator
def evaluate(
self,
evo: FactorImplementationList,
queried_knowledge: QueriedKnowledge = None,
**kwargs,
) -> FactorImplementationsMultiFeedback:
multi_implementation_feedback = FactorImplementationsMultiFeedback()
# for index in range(len(evo.target_factor_tasks)):
# corresponding_implementation = evo.corresponding_implementations[index]
# corresponding_gt_implementation = (
# evo.corresponding_gt_implementations[index]
# if evo.corresponding_gt_implementations is not None
# else None
# )
# multi_implementation_feedback.append(
# self.single_factor_implementation_evaluator.evaluate(
# target_task=evo.target_factor_tasks[index],
# implementation=corresponding_implementation,
# gt_implementation=corresponding_gt_implementation,
# queried_knowledge=queried_knowledge,
# )
# )
calls = []
for index in range(len(evo.target_factor_tasks)):
corresponding_implementation = evo.corresponding_implementations[index]
corresponding_gt_implementation = (
evo.corresponding_gt_implementations[index]
if evo.corresponding_gt_implementations is not None
else None
)
calls.append(
(
self.single_factor_implementation_evaluator.evaluate,
(
evo.target_factor_tasks[index],
corresponding_implementation,
corresponding_gt_implementation,
queried_knowledge,
),
),
)
multi_implementation_feedback = multiprocessing_wrapper(calls, n=FactorImplementSettings().evo_multi_proc_n)
final_decision = [
None if single_feedback is None else single_feedback.final_decision
for single_feedback in multi_implementation_feedback
]
print(f"Final decisions: {final_decision} True count: {final_decision.count(True)}")
return multi_implementation_feedback
@@ -0,0 +1,34 @@
from __future__ import annotations
import pandas as pd
from core.evolving_framework import EvolvableSubjects
from core.log import FinCoLog
from factor_implementation.share_modules.factor import (
FactorImplementation,
FactorImplementationTask,
)
class FactorImplementationList(EvolvableSubjects):
"""
Factors is a list.
"""
def __init__(
self,
target_factor_tasks: list[FactorImplementationTask],
corresponding_gt_implementations: list[FactorImplementation] = None,
):
super().__init__()
self.target_factor_tasks = target_factor_tasks
self.corresponding_implementations: list[FactorImplementation] = []
if corresponding_gt_implementations is not None and len(
corresponding_gt_implementations,
) != len(target_factor_tasks):
self.corresponding_gt_implementations = None
FinCoLog.warning(
"The length of corresponding_gt_implementations is not equal to the length of target_factor_tasks, set corresponding_gt_implementations to None",
)
else:
self.corresponding_gt_implementations = corresponding_gt_implementations
@@ -0,0 +1,298 @@
from __future__ import annotations
import json
import random
from abc import abstractmethod
from copy import deepcopy
from typing import TYPE_CHECKING
from jinja2 import Template
from core.evolving_framework import EvolvingStrategy, QueriedKnowledge
from oai.llm_utils import APIBackend
from factor_implementation.share_modules.conf import FactorImplementSettings
from factor_implementation.share_modules.factor import (
FactorImplementation,
FactorImplementationTask,
FileBasedFactorImplementation,
)
from factor_implementation.share_modules.prompt import (
FactorImplementationPrompts,
)
from factor_implementation.share_modules.utils import get_data_folder_intro
from core.utils import multiprocessing_wrapper
if TYPE_CHECKING:
from factor_implementation.evolving.evolvable_subjects import (
FactorImplementationList,
)
from factor_implementation.evolving.knowledge_management import (
FactorImplementationQueriedKnowledge,
FactorImplementationQueriedKnowledgeV1,
)
class MultiProcessEvolvingStrategy(EvolvingStrategy):
@abstractmethod
def implement_one_factor(
self,
target_task: FactorImplementationTask,
queried_knowledge: QueriedKnowledge = None,
) -> FactorImplementation:
raise NotImplementedError
def evolve(
self,
*,
evo: FactorImplementationList,
queried_knowledge: FactorImplementationQueriedKnowledge | None = None,
**kwargs,
) -> FactorImplementationList:
new_evo = deepcopy(evo)
new_evo.corresponding_implementations = [None for _ in new_evo.target_factor_tasks]
to_be_finished_task_index = []
for index, target_factor_task in enumerate(new_evo.target_factor_tasks):
target_factor_task_desc = target_factor_task.get_factor_information()
if target_factor_task_desc in queried_knowledge.success_task_to_knowledge_dict:
new_evo.corresponding_implementations[index] = queried_knowledge.success_task_to_knowledge_dict[
target_factor_task_desc
].implementation
elif (
target_factor_task_desc not in queried_knowledge.success_task_to_knowledge_dict
and target_factor_task_desc not in queried_knowledge.failed_task_info_set
):
to_be_finished_task_index.append(index)
if FactorImplementSettings().implementation_factors_per_round < len(to_be_finished_task_index):
to_be_finished_task_index = random.sample(
to_be_finished_task_index,
FactorImplementSettings().implementation_factors_per_round,
)
result = multiprocessing_wrapper(
[
(self.implement_one_factor, (new_evo.target_factor_tasks[target_index], queried_knowledge))
for target_index in to_be_finished_task_index
],
n=FactorImplementSettings().evo_multi_proc_n,
)
for index, target_index in enumerate(to_be_finished_task_index):
new_evo.corresponding_implementations[target_index] = result[index]
# for target_index in to_be_finished_task_index:
# new_evo.corresponding_implementations[target_index] = self.implement_one_factor(
# new_evo.target_factor_tasks[target_index], queried_knowledge
# )
return new_evo
class FactorEvolvingStrategy(MultiProcessEvolvingStrategy):
def implement_one_factor(
self,
target_task: FactorImplementationTask,
queried_knowledge: FactorImplementationQueriedKnowledgeV1 = None,
) -> FactorImplementation:
factor_information_str = target_task.get_factor_information()
if queried_knowledge is not None and factor_information_str in queried_knowledge.success_task_to_knowledge_dict:
return queried_knowledge.success_task_to_knowledge_dict[factor_information_str].implementation
elif queried_knowledge is not None and factor_information_str in queried_knowledge.failed_task_info_set:
return None
else:
queried_similar_successful_knowledge = (
queried_knowledge.working_task_to_similar_successful_knowledge_dict[factor_information_str]
if queried_knowledge is not None
else []
)
queried_former_failed_knowledge = (
queried_knowledge.working_task_to_former_failed_knowledge_dict[factor_information_str]
if queried_knowledge is not None
else []
)
queried_former_failed_knowledge_to_render = queried_former_failed_knowledge
system_prompt = Template(
FactorImplementationPrompts()["evolving_strategy_factor_implementation_v1_system"],
).render(
data_info=get_data_folder_intro(),
queried_former_failed_knowledge=queried_former_failed_knowledge_to_render,
)
session = APIBackend(use_chat_cache=False).build_chat_session(
session_system_prompt=system_prompt,
)
queried_similar_successful_knowledge_to_render = queried_similar_successful_knowledge
while True:
user_prompt = (
Template(
FactorImplementationPrompts()["evolving_strategy_factor_implementation_v1_user"],
)
.render(
factor_information_str=factor_information_str,
queried_similar_successful_knowledge=queried_similar_successful_knowledge_to_render,
)
.strip("\n")
)
if (
session.build_chat_completion_message_and_calculate_token(
user_prompt,
)
< FactorImplementSettings().chat_token_limit
):
break
elif len(queried_former_failed_knowledge_to_render) > 1:
queried_former_failed_knowledge_to_render = queried_former_failed_knowledge_to_render[1:]
elif len(queried_similar_successful_knowledge_to_render) > 1:
queried_similar_successful_knowledge_to_render = queried_similar_successful_knowledge_to_render[1:]
# print(
# f"length of queried_similar_successful_knowledge_to_render: {len(queried_similar_successful_knowledge_to_render)}, length of queried_former_failed_knowledge_to_render: {len(queried_former_failed_knowledge_to_render)}"
# )
code = json.loads(
session.build_chat_completion(
user_prompt=user_prompt,
json_mode=True,
),
)["code"]
# ast.parse(code)
factor_implementation = FileBasedFactorImplementation(
target_task,
code,
)
return factor_implementation
class FactorEvolvingStrategyWithGraph(MultiProcessEvolvingStrategy):
def implement_one_factor(
self,
target_task: FactorImplementationTask,
queried_knowledge,
) -> FactorImplementation:
error_summary = FactorImplementSettings().v2_error_summary
target_factor_task_information = target_task.get_factor_information()
if (
queried_knowledge is not None
and target_factor_task_information in queried_knowledge.success_task_to_knowledge_dict
):
return queried_knowledge.success_task_to_knowledge_dict[target_factor_task_information].implementation
elif queried_knowledge is not None and target_factor_task_information in queried_knowledge.failed_task_info_set:
return None
else:
queried_similar_component_knowledge = (
queried_knowledge.component_with_success_task[target_factor_task_information]
if queried_knowledge is not None
else []
) # A list, [success task implement knowledge]
queried_similar_error_knowledge = (
queried_knowledge.error_with_success_task[target_factor_task_information]
if queried_knowledge is not None
else {}
) # A dict, {{error_type:[[error_imp_knowledge, success_imp_knowledge],...]},...}
queried_former_failed_knowledge = (
queried_knowledge.former_traces[target_factor_task_information] if queried_knowledge is not None else []
)
queried_former_failed_knowledge_to_render = queried_former_failed_knowledge
system_prompt = Template(
FactorImplementationPrompts()["evolving_strategy_factor_implementation_v1_system"],
).render(
data_info=get_data_folder_intro(),
queried_former_failed_knowledge=queried_former_failed_knowledge_to_render,
)
session = APIBackend(use_chat_cache=False).build_chat_session(
session_system_prompt=system_prompt,
)
queried_similar_component_knowledge_to_render = queried_similar_component_knowledge
queried_similar_error_knowledge_to_render = queried_similar_error_knowledge
error_summary_critics = ""
while True:
if (
error_summary
and len(queried_similar_error_knowledge_to_render) != 0
and len(queried_former_failed_knowledge_to_render) != 0
):
error_summary_system_prompt = (
Template(FactorImplementationPrompts()["evolving_strategy_error_summary_v2_system"])
.render(
factor_information_str=target_factor_task_information,
code_and_feedback=queried_former_failed_knowledge_to_render[
-1
].get_implementation_and_feedback_str(),
)
.strip("\n")
)
session_summary = APIBackend(use_chat_cache=False).build_chat_session(
session_system_prompt=error_summary_system_prompt,
)
while True:
error_summary_user_prompt = (
Template(FactorImplementationPrompts()["evolving_strategy_error_summary_v2_user"])
.render(
queried_similar_component_knowledge=queried_similar_component_knowledge_to_render,
)
.strip("\n")
)
if (
session_summary.build_chat_completion_message_and_calculate_token(error_summary_user_prompt)
< FactorImplementSettings().chat_token_limit
):
break
elif len(queried_similar_error_knowledge_to_render) > 0:
queried_similar_error_knowledge_to_render = queried_similar_error_knowledge_to_render[:-1]
error_summary_critics = session_summary.build_chat_completion(
user_prompt=error_summary_user_prompt,
json_mode=False,
)
user_prompt = (
Template(
FactorImplementationPrompts()["evolving_strategy_factor_implementation_v2_user"],
)
.render(
factor_information_str=target_factor_task_information,
queried_similar_component_knowledge=queried_similar_component_knowledge_to_render,
queried_similar_error_knowledge=queried_similar_error_knowledge_to_render,
error_summary=error_summary,
error_summary_critics=error_summary_critics,
)
.strip("\n")
)
if (
session.build_chat_completion_message_and_calculate_token(
user_prompt,
)
< FactorImplementSettings().chat_token_limit
):
break
elif len(queried_former_failed_knowledge_to_render) > 1:
queried_former_failed_knowledge_to_render = queried_former_failed_knowledge_to_render[1:]
elif len(queried_similar_component_knowledge_to_render) > len(
queried_similar_error_knowledge_to_render,
):
queried_similar_component_knowledge_to_render = queried_similar_component_knowledge_to_render[:-1]
elif len(queried_similar_error_knowledge_to_render) > 0:
queried_similar_error_knowledge_to_render = queried_similar_error_knowledge_to_render[:-1]
# print(
# len(queried_similar_component_knowledge_to_render),
# len(queried_similar_error_knowledge_to_render),
# len(queried_former_failed_knowledge_to_render),
# )
response = session.build_chat_completion(
user_prompt=user_prompt,
json_mode=True,
)
code = json.loads(response)["code"]
factor_implementation = FileBasedFactorImplementation(target_task, code)
return factor_implementation
@@ -0,0 +1,311 @@
import json
import pickle
import subprocess
from pathlib import Path
import pandas as pd
from fire.core import Fire
from tqdm import tqdm
from core.evolving_framework import EvoAgent, KnowledgeBase
from factor_implementation.evolving.evaluators import (
FactorImplementationEvaluatorV1,
FactorImplementationsMultiEvaluator,
)
from factor_implementation.evolving.evolvable_subjects import (
FactorImplementationList,
)
from factor_implementation.evolving.evolving_strategy import (
FactorEvolvingStrategy,
FactorEvolvingStrategyWithGraph,
)
from factor_implementation.evolving.knowledge_management import (
FactorImplementationGraphKnowledgeBase,
FactorImplementationGraphRAGStrategy,
FactorImplementationKnowledgeBaseV1,
FactorImplementationRAGStrategyV1,
)
from factor_implementation.share_modules.factor import (
FactorImplementationTask,
FileBasedFactorImplementation,
)
from core.utils import multiprocessing_wrapper
ALPHA101_INIT_COMPONENTS = [
"1. abs(): absolute value to certain columns",
"2. log(): log value to certain columns",
"3. sign(): sign value to certain columns",
"4. add_two_columns(): add two columns",
"5. minus_two_columns(): minus two columns",
"6. times_two_columns(): times two columns",
"7. divide_two_columns(): divide two columns",
"8. add_value_to_columns(): add value to columns",
"9. minus_value_to_columns(): minus value to columns",
"10. rank(): cross-sectional rank value to columns",
"11. delay(): value of each data d days ago",
"12. correlation(): time-serial correlation of column_left and column_right for the past d days",
"13. covariance(): time-serial covariance of column_left and column_right for the past d days",
"14. scale_to_a(): scale the columns to sum(abs(x)) is a",
"15. delta(): todays value of x minus the value of x d days ago",
"16. signedpower(): x^a",
"17. decay_linear(): weighted moving average over the past d days with linearly decaying weights d, d 1, …, 1 (rescaled to sum up to 1)",
"18. indneutralize(): x cross-sectionally neutralized against groups g (subindustries, industries, sectors, etc.), i.e., x is cross-sectionally demeaned within each group g",
"19. ts_min(): time-series min over the past d days, operator min applied across the time-series for the past d days; non-integer number of days d is converted to floor(d)",
"20. ts_max(): time-series max over the past d days, operator max applied across the time-series for the past d days; non-integer number of days d is converted to floor(d)",
"21. ts_argmax(): which day ts_max(x, d) occurred on",
"22. ts_argmin(): which day ts_min(x, d) occurred on",
"23. ts_rank(): time-series rank in the past d days",
"24. min(): ts_min(x, d)",
"25. max(): ts_max(x, d)",
"26. sum(): time-series sum over the past d days",
"27. product(): time-series product over the past d days",
"28. stddev(): moving time-series standard deviation over the past d days",
]
class FactorImplementationEvolvingCli:
# TODO: we should use polymorphism to load knowledge base, strategies instead of evolving_version
# TODO: Can we refactor FactorImplementationEvolvingCli into a learning framework to differentiate our learning paradiagm with other ones by iteratively retrying?
def __init__(self, evolving_version=2) -> None:
self.evolving_version = evolving_version
self.knowledge_base = None
self.latest_factor_implementations = None
def run_evolving_framework(
self,
factor_implementations: FactorImplementationList,
factor_knowledge_base: KnowledgeBase,
max_loops: int = 20,
with_knowledge: bool = True,
with_feedback: bool = True,
knowledge_self_gen: bool = True,
) -> FactorImplementationList:
"""
Main target: Implement factors.
The system also leverages the former knowledge to help implement the factors. Also, new knowledge might be generated during the implementation to help the following implementation.
The gt_code and gt_value in the Factor instance is used to evaluate the implementation, and the feedback is used to generate high-quality knowledge which helps the agent to evolve.
"""
es = FactorEvolvingStrategyWithGraph() if self.evolving_version == 2 else FactorEvolvingStrategy()
rag = (
FactorImplementationGraphRAGStrategy(factor_knowledge_base)
if self.evolving_version == 2
else FactorImplementationRAGStrategyV1(factor_knowledge_base)
)
factor_evaluator = FactorImplementationsMultiEvaluator(FactorImplementationEvaluatorV1())
ea = EvoAgent(es, rag=rag)
for _ in tqdm(range(max_loops), "Implementing factors"):
factor_implementations = ea.step_evolving(
factor_implementations,
factor_evaluator,
with_knowledge=with_knowledge,
with_feedback=with_feedback,
knowledge_self_gen=knowledge_self_gen,
)
return factor_implementations
def load_or_init_knowledge_base(self, former_knowledge_base_path: Path = None, component_init_list: list = []):
if former_knowledge_base_path is not None and former_knowledge_base_path.exists():
factor_knowledge_base = pickle.load(open(former_knowledge_base_path, "rb"))
if self.evolving_version == 1 and not isinstance(
factor_knowledge_base, FactorImplementationKnowledgeBaseV1
):
raise ValueError("The former knowledge base is not compatible with the current version")
elif self.evolving_version == 2 and not isinstance(
factor_knowledge_base,
FactorImplementationGraphKnowledgeBase,
):
raise ValueError("The former knowledge base is not compatible with the current version")
else:
factor_knowledge_base = (
FactorImplementationGraphKnowledgeBase(
init_component_list=component_init_list,
)
if self.evolving_version == 2
else FactorImplementationKnowledgeBaseV1()
)
return factor_knowledge_base
def implement_factors(
self,
factor_implementations: FactorImplementationList,
former_knowledge_base_path: Path = None,
new_knowledge_base_path: Path = None,
component_init_list: list = [],
max_loops: int = 20,
):
factor_knowledge_base = self.load_or_init_knowledge_base(
former_knowledge_base_path=former_knowledge_base_path,
component_init_list=component_init_list,
)
new_factor_implementations = self.run_evolving_framework(
factor_implementations=factor_implementations,
factor_knowledge_base=factor_knowledge_base,
max_loops=max_loops,
with_knowledge=True,
with_feedback=True,
knowledge_self_gen=True,
)
if new_knowledge_base_path is not None:
pickle.dump(factor_knowledge_base, open(new_knowledge_base_path, "wb"))
self.knowledge_base = factor_knowledge_base
self.latest_factor_implementations = factor_implementations
return new_factor_implementations
def _read_alpha101_factors(
self,
alpha101_evo_subs_path: Path = None,
alpha101_data_path=Path().cwd() / "git_ignore_folder" / "alpha101_related_files",
start_index=0,
end_index=32,
read_gt_factors=True,
) -> FactorImplementationList:
"""
Read the alpha101 factors from the alpha101_related_files folder
"""
if alpha101_evo_subs_path is not None and alpha101_evo_subs_path.exists():
factor_implementations = pickle.load(open(alpha101_evo_subs_path, "rb"))
else:
target_factor_plain_list = json.load(
open(alpha101_data_path / "target_factor_task_list.json"),
)
name_to_code = json.load(open(alpha101_data_path / "name_to_code.json"))
gt_df = pd.read_hdf(alpha101_data_path / "gt_filtered.h5")
# First read the target factor task
target_factor_tasks = []
for factor_list_item in target_factor_plain_list:
target_factor_tasks.append(
FactorImplementationTask(
factor_name=factor_list_item[0],
factor_description=factor_list_item[1],
factor_formulation=factor_list_item[2],
factor_formulation_description=factor_list_item[3],
),
)
# Second read the gt factor implementations
corresponding_gt_implementations = []
for factor_task in target_factor_tasks:
name = factor_task.factor_name
gt_code = name_to_code[name]
gt_value = gt_df.loc(axis=1)[[name]]
corresponding_gt_implementations.append(
FileBasedFactorImplementation(
code=gt_code,
executed_factor_value_dataframe=gt_value,
target_task=factor_task,
),
)
# Finally generate the factor implementations as evolvable subjects
factor_implementations = FactorImplementationList(
target_factor_tasks=target_factor_tasks,
corresponding_gt_implementations=(corresponding_gt_implementations if read_gt_factors else None),
)
factor_implementations.target_factor_tasks = factor_implementations.target_factor_tasks[start_index:end_index]
factor_implementations.corresponding_gt_implementations = (
factor_implementations.corresponding_gt_implementations[start_index:end_index] if read_gt_factors else None
)
return factor_implementations
def implement_alpha101(
self,
max_loops=30,
) -> FactorImplementationList:
"""
Implement the alpha101 factors to gather knowledge TODO: implement the code
"""
factor_implementations = self._read_alpha101_factors(
alpha101_evo_subs_path=Path.cwd() / "alpha101_evo_subs.pkl",
start_index=0,
end_index=64,
read_gt_factors=True,
)
self.implement_factors(
factor_implementations,
former_knowledge_base_path=Path.cwd()
/ f"alpha101_knowledge_base_v{self.evolving_version}_project_product.pkl",
new_knowledge_base_path=Path.cwd()
/ f"alpha101_knowledge_base_v{self.evolving_version}_project_product.pkl",
component_init_list=ALPHA101_INIT_COMPONENTS,
max_loops=100,
)
factor_implementations = self._read_alpha101_factors(
alpha101_evo_subs_path=Path.cwd() / "alpha101_evo_subs.pkl",
start_index=64,
end_index=96,
read_gt_factors=False,
)
final_imp = self.implement_factors(
factor_implementations,
former_knowledge_base_path=Path.cwd()
/ f"alpha101_knowledge_base_v{self.evolving_version}_project_product.pkl",
new_knowledge_base_path=Path.cwd()
/ f"alpha101_knowledge_base_v{self.evolving_version}_self_evolving_project_product.pkl",
component_init_list=ALPHA101_INIT_COMPONENTS,
max_loops=10,
)
final_imp.corresponding_gt_implementations = factor_implementations = self._read_alpha101_factors(
alpha101_evo_subs_path=Path.cwd() / "alpha101_evo_subs.pkl",
start_index=64,
end_index=96,
read_gt_factors=True,
).corresponding_gt_implementations
feedbacks = FactorImplementationsMultiEvaluator().evaluate(final_imp)
print([feedback.final_decision if feedback is not None else None for feedback in feedbacks].count(True))
def implement_amc(
self, evo_sub_path_str, former_knowledge_base_path_str, implementation_dump_path_str, slice_index
):
factor_implementations: FactorImplementationList = pickle.load(open(evo_sub_path_str, "rb"))
factor_implementations.target_factor_tasks = factor_implementations.target_factor_tasks[
slice_index * 16 : slice_index * 16 + 16
]
if len(factor_implementations.target_factor_tasks) == 0:
return
if Path(implementation_dump_path_str).exists():
return
factor_implementations = self.implement_factors(
factor_implementations,
former_knowledge_base_path=Path(former_knowledge_base_path_str),
component_init_list=ALPHA101_INIT_COMPONENTS,
max_loops=10,
)
pickle.dump(factor_implementations, open(implementation_dump_path_str, "wb"))
def execute_command(self, command, cwd):
print(command, cwd)
try:
subprocess.check_output(
command,
shell=True,
cwd=cwd,
)
except subprocess.CalledProcessError as e:
print(e.output.decode())
def multi_inference_amc_factors(self, type):
slice_count = {"price_volume": 35, "fundamental": 24, "high_frequency": 16}[type]
res = multiprocessing_wrapper(
[
(
self.execute_command,
(
f"python src/scripts/factor_implementation/baselines/evolving/factor_implementation_evolving_cli.py implement_amc ./{type}_factors.pkl ./knowledge_base_v2_with_alpha101_and_10_factors.pkl ./inference_amc_factors_{type}_{slice}.pkl {slice}",
Path.cwd(),
),
)
for slice in range(slice_count)
],
n=2,
)
if __name__ == "__main__":
Fire(FactorImplementationEvolvingCli)
@@ -0,0 +1,905 @@
from __future__ import annotations
import copy
import json
import random
import re
from itertools import combinations
from pathlib import Path
from typing import Union
from jinja2 import Template
from core.evolving_framework import (
EvolvableSubjects,
EvoStep,
Knowledge,
KnowledgeBase,
QueriedKnowledge,
RAGStrategy,
)
from finco.graph import UndirectedGraph, UndirectedNode
from oai.llm_utils import APIBackend, calculate_embedding_distance_between_str_list
from core.log import FinCoLog
from factor_implementation.evolving.evaluators import (
FactorImplementationSingleFeedback,
)
from factor_implementation.share_modules.conf import FactorImplementSettings
from factor_implementation.share_modules.factor import (
FactorImplementation,
FactorImplementationTask,
)
from factor_implementation.share_modules.prompt import (
FactorImplementationPrompts,
)
class FactorImplementationKnowledge(Knowledge):
def __init__(
self,
target_task: FactorImplementationTask,
implementation: FactorImplementation,
feedback: FactorImplementationSingleFeedback,
) -> None:
"""
Initialize a FactorKnowledge object. The FactorKnowledge object is used to store a factor implementation without the ground truth code and value.
Args:
factor (Factor): The factor object associated with the KnowledgeManagement.
Returns:
None
"""
self.target_task = target_task
self.implementation = implementation
self.feedback = feedback
def get_implementation_and_feedback_str(self) -> str:
return f"""------------------Factor implementation code:------------------
{self.implementation.code}
------------------Factor implementation feedback:------------------
{self.feedback!s}
"""
class FactorImplementationQueriedKnowledge(QueriedKnowledge):
def __init__(self, success_task_to_knowledge_dict: dict = {}, failed_task_info_set: set = set()) -> None:
self.success_task_to_knowledge_dict = success_task_to_knowledge_dict
self.failed_task_info_set = failed_task_info_set
class FactorImplementationKnowledgeBaseV1(KnowledgeBase):
def __init__(self) -> None:
self.implementation_trace: dict[str, FactorImplementationKnowledge] = dict()
self.success_task_info_set: set[str] = set()
self.task_to_embedding = dict()
def query(self) -> QueriedKnowledge | None:
"""
Query the knowledge base to get the queried knowledge. So far is handled in RAG strategy.
"""
raise NotImplementedError
class FactorImplementationQueriedKnowledgeV1(FactorImplementationQueriedKnowledge):
def __init__(self) -> None:
self.working_task_to_former_failed_knowledge_dict = dict()
self.working_task_to_similar_successful_knowledge_dict = dict()
super().__init__()
class FactorImplementationRAGStrategyV1(RAGStrategy):
def __init__(self, knowledgebase: FactorImplementationKnowledgeBaseV1) -> None:
super().__init__(knowledgebase)
self.current_generated_trace_count = 0
def generate_knowledge(
self,
evolving_trace: list[EvoStep],
*,
return_knowledge: bool = False,
) -> Knowledge | None:
if len(evolving_trace) == self.current_generated_trace_count:
return
else:
for trace_index in range(
self.current_generated_trace_count,
len(evolving_trace),
):
evo_step = evolving_trace[trace_index]
implementations = evo_step.evolvable_subjects
feedback = evo_step.feedback
for task_index in range(len(implementations.target_factor_tasks)):
target_task = implementations.target_factor_tasks[task_index]
target_task_information = target_task.get_factor_information()
implementation = implementations.corresponding_implementations[task_index]
single_feedback = feedback[task_index]
if single_feedback is None:
continue
single_knowledge = FactorImplementationKnowledge(
target_task=target_task,
implementation=implementation,
feedback=single_feedback,
)
if target_task_information not in self.knowledgebase.success_task_info_set:
self.knowledgebase.implementation_trace.setdefault(
target_task_information,
[],
).append(single_knowledge)
if single_feedback.final_decision == True:
self.knowledgebase.success_task_info_set.add(
target_task_information,
)
self.current_generated_trace_count = len(evolving_trace)
def query(
self,
evo: EvolvableSubjects,
evolving_trace: list[EvoStep],
) -> QueriedKnowledge | None:
v1_query_former_trace_limit = FactorImplementSettings().v1_query_former_trace_limit
v1_query_similar_success_limit = FactorImplementSettings().v1_query_similar_success_limit
fail_task_trial_limit = FactorImplementSettings().fail_task_trial_limit
queried_knowledge = FactorImplementationQueriedKnowledgeV1()
for target_factor_task in evo.target_factor_tasks:
target_factor_task_information = target_factor_task.get_factor_information()
if target_factor_task_information in self.knowledgebase.success_task_info_set:
queried_knowledge.success_task_to_knowledge_dict[target_factor_task_information] = (
self.knowledgebase.implementation_trace[target_factor_task_information][-1]
)
else:
if (
len(
self.knowledgebase.implementation_trace.setdefault(
target_factor_task_information,
[],
),
)
>= fail_task_trial_limit
):
queried_knowledge.failed_task_info_set.add(target_factor_task_information)
else:
queried_knowledge.working_task_to_former_failed_knowledge_dict[target_factor_task_information] = (
self.knowledgebase.implementation_trace.setdefault(
target_factor_task_information,
[],
)[-v1_query_former_trace_limit:]
)
knowledge_base_success_task_list = list(
self.knowledgebase.success_task_info_set,
)
similarity = calculate_embedding_distance_between_str_list(
[target_factor_task_information],
knowledge_base_success_task_list,
)[0]
similar_indexes = sorted(
range(len(similarity)),
key=lambda i: similarity[i],
reverse=True,
)[:v1_query_similar_success_limit]
similar_successful_knowledge = [
self.knowledgebase.implementation_trace.setdefault(
knowledge_base_success_task_list[index],
[],
)[-1]
for index in similar_indexes
]
queried_knowledge.working_task_to_similar_successful_knowledge_dict[
target_factor_task_information
] = similar_successful_knowledge
return queried_knowledge
class FactorImplementationQueriedGraphKnowledge(FactorImplementationQueriedKnowledge):
# Aggregation of knowledge
def __init__(
self,
former_traces: dict = {},
component_with_success_task: dict = {},
error_with_success_task: dict = {},
**kwargs,
) -> None:
self.former_traces = former_traces
self.component_with_success_task = component_with_success_task
self.error_with_success_task = error_with_success_task
super().__init__(**kwargs)
class FactorImplementationGraphRAGStrategy(RAGStrategy):
def __init__(self, knowledgebase: FactorImplementationGraphKnowledgeBase) -> None:
super().__init__(knowledgebase)
self.current_generated_trace_count = 0
self.prompt = FactorImplementationPrompts()
def generate_knowledge(
self,
evolving_trace: list[EvoStep],
*,
return_knowledge: bool = False,
) -> Knowledge | None:
if len(evolving_trace) == self.current_generated_trace_count:
return None
else:
for trace_index in range(self.current_generated_trace_count, len(evolving_trace)):
evo_step = evolving_trace[trace_index]
implementations = evo_step.evolvable_subjects
feedback = evo_step.feedback
for task_index in range(len(implementations.target_factor_tasks)):
single_feedback = feedback[task_index]
target_task = implementations.target_factor_tasks[task_index]
target_task_information = target_task.get_factor_information()
implementation = implementations.corresponding_implementations[task_index]
single_feedback = feedback[task_index]
if single_feedback is None:
continue
single_knowledge = FactorImplementationKnowledge(
target_task=target_task,
implementation=implementation,
feedback=single_feedback,
)
if (
target_task_information not in self.knowledgebase.success_task_to_knowledge_dict
and implementation is not None
):
self.knowledgebase.working_trace_knowledge.setdefault(target_task_information, []).append(
single_knowledge,
) # save to working trace
if single_feedback.final_decision == True:
self.knowledgebase.success_task_to_knowledge_dict.setdefault(
target_task_information,
single_knowledge,
)
# Do summary for the last step and update the knowledge graph
self.knowledgebase.update_success_task(
target_task_information,
)
else:
# generate error node and store into knowledge base
error_analysis_result = []
if not single_feedback.value_generated_flag:
error_analysis_result = self.analyze_error(
single_feedback.execution_feedback,
feedback_type="execution",
)
else:
error_analysis_result = self.analyze_error(
single_feedback.factor_value_feedback,
feedback_type="value",
)
self.knowledgebase.working_trace_error_analysis.setdefault(
target_task_information,
[],
).append(
error_analysis_result,
) # save to working trace error record, for graph update
self.current_generated_trace_count = len(evolving_trace)
return None
def query(self, evo: EvolvableSubjects, evolving_trace: list[EvoStep]) -> QueriedKnowledge | None:
conf_knowledge_sampler = FactorImplementSettings().v2_knowledge_sampler
factor_implementation_queried_graph_knowledge = FactorImplementationQueriedGraphKnowledge(
success_task_to_knowledge_dict=self.knowledgebase.success_task_to_knowledge_dict,
)
factor_implementation_queried_graph_knowledge = self.former_trace_query(
evo,
factor_implementation_queried_graph_knowledge,
FactorImplementSettings().v2_query_former_trace_limit,
)
factor_implementation_queried_graph_knowledge = self.component_query(
evo,
factor_implementation_queried_graph_knowledge,
FactorImplementSettings().v2_query_component_limit,
knowledge_sampler=conf_knowledge_sampler,
)
factor_implementation_queried_graph_knowledge = self.error_query(
evo,
factor_implementation_queried_graph_knowledge,
FactorImplementSettings().v2_query_error_limit,
knowledge_sampler=conf_knowledge_sampler,
)
return factor_implementation_queried_graph_knowledge
def analyze_component(
self,
target_factor_task_information,
) -> list[UndirectedNode]: # Hardcode: certain component nodes
all_component_nodes = self.knowledgebase.graph.get_all_nodes_by_label_list(["component"])
all_component_content = ""
for _, component_node in enumerate(all_component_nodes):
all_component_content += f"{component_node.content}, \n"
analyze_component_system_prompt = Template(self.prompt["analyze_component_prompt_v1_system"]).render(
all_component_content=all_component_content,
)
analyze_component_user_prompt = target_factor_task_information
try:
component_no_list = json.loads(
APIBackend().build_messages_and_create_chat_completion(
system_prompt=analyze_component_system_prompt,
user_prompt=analyze_component_user_prompt,
json_mode=True,
),
)["component_no_list"]
return [all_component_nodes[index - 1] for index in sorted(list(set(component_no_list)))]
except:
FinCoLog.warning("Error when analyzing components.")
analyze_component_user_prompt = "Your response is not a valid component index list."
return []
def analyze_error(
self,
single_feedback,
feedback_type="execution",
) -> list[
UndirectedNode | str
]: # Hardcode: Raised errors, existed error nodes + not existed error nodes(here, they are strs)
if feedback_type == "execution":
match = re.search(
r'File "(?P<file>.+)", line (?P<line>\d+), in (?P<function>.+)\n\s+(?P<error_line>.+)\n(?P<error_type>\w+): (?P<error_message>.+)',
single_feedback,
)
if match:
error_details = match.groupdict()
# last_traceback = f'File "{error_details["file"]}", line {error_details["line"]}, in {error_details["function"]}\n {error_details["error_line"]}'
error_type = error_details["error_type"]
error_line = error_details["error_line"]
error_contents = [f"ErrorType: {error_type}" + "\n" + f"Error line: {error_line}"]
else:
error_contents = ["Undefined Error"]
elif feedback_type == "value": # value check error
value_check_types = r"The source dataframe and the ground truth dataframe have different rows count.|The source dataframe and the ground truth dataframe have different index.|Some values differ by more than the tolerance of 1e-6.|No sufficient correlation found when shifting up|Something wrong happens when naming the multi indices of the dataframe."
error_contents = re.findall(value_check_types, single_feedback)
else:
error_contents = ["Undefined Error"]
all_error_nodes = self.knowledgebase.graph.get_all_nodes_by_label_list(["error"])
if not len(all_error_nodes):
return error_contents
else:
error_list = []
for error_content in error_contents:
for error_node in all_error_nodes:
if error_content == error_node.content:
error_list.append(error_node)
else:
error_list.append(error_content)
if error_list[-1] in error_list[:-1]:
error_list.pop()
return error_list
def former_trace_query(
self,
evo: EvolvableSubjects,
factor_implementation_queried_graph_knowledge: FactorImplementationQueriedGraphKnowledge,
v2_query_former_trace_limit: int = 5,
) -> Union[QueriedKnowledge, set]:
"""
Query the former trace knowledge of the working trace, and find all the failed task information which tried more than fail_task_trial_limit times
"""
fail_task_trial_limit = FactorImplementSettings().fail_task_trial_limit
for target_factor_task in evo.target_factor_tasks:
target_factor_task_information = target_factor_task.get_factor_information()
if (
target_factor_task_information not in self.knowledgebase.success_task_to_knowledge_dict
and target_factor_task_information in self.knowledgebase.working_trace_knowledge
and len(self.knowledgebase.working_trace_knowledge[target_factor_task_information])
>= fail_task_trial_limit
):
factor_implementation_queried_graph_knowledge.failed_task_info_set.add(target_factor_task_information)
if (
target_factor_task_information not in self.knowledgebase.success_task_to_knowledge_dict
and target_factor_task_information
not in factor_implementation_queried_graph_knowledge.failed_task_info_set
and target_factor_task_information in self.knowledgebase.working_trace_knowledge
):
former_trace_knowledge = copy.copy(
self.knowledgebase.working_trace_knowledge[target_factor_task_information],
)
# in former trace query we will delete the right trace in the following order:[..., value_generated_flag is True, value_generated_flag is False, ...]
# because we think this order means a deterioration of the trial (like a wrong gradient descent)
current_index = 1
while current_index < len(former_trace_knowledge):
if (
not former_trace_knowledge[current_index].feedback.value_generated_flag
and former_trace_knowledge[current_index - 1].feedback.value_generated_flag
):
former_trace_knowledge.pop(current_index)
else:
current_index += 1
factor_implementation_queried_graph_knowledge.former_traces[target_factor_task_information] = (
former_trace_knowledge[-v2_query_former_trace_limit:]
)
else:
factor_implementation_queried_graph_knowledge.former_traces[target_factor_task_information] = []
return factor_implementation_queried_graph_knowledge
def component_query(
self,
evo: EvolvableSubjects,
factor_implementation_queried_graph_knowledge: FactorImplementationQueriedGraphKnowledge,
v2_query_component_limit: int = 5,
knowledge_sampler: float = 1.0,
) -> QueriedKnowledge | None:
# queried_component_knowledge = FactorImplementationQueriedGraphComponentKnowledge()
for target_factor_task in evo.target_factor_tasks:
target_factor_task_information = target_factor_task.get_factor_information()
if (
target_factor_task_information in self.knowledgebase.success_task_to_knowledge_dict
or target_factor_task_information in factor_implementation_queried_graph_knowledge.failed_task_info_set
):
factor_implementation_queried_graph_knowledge.component_with_success_task[
target_factor_task_information
] = []
else:
if target_factor_task_information not in self.knowledgebase.task_to_component_nodes:
self.knowledgebase.task_to_component_nodes[target_factor_task_information] = self.analyze_component(
target_factor_task_information,
)
component_analysis_result = self.knowledgebase.task_to_component_nodes[target_factor_task_information]
if len(component_analysis_result) > 1:
task_des_node_list = self.knowledgebase.graph_query_by_intersection(
component_analysis_result,
constraint_labels=["task_description"],
)
single_component_constraint = (v2_query_component_limit // len(component_analysis_result)) + 1
else:
task_des_node_list = []
single_component_constraint = v2_query_component_limit
factor_implementation_queried_graph_knowledge.component_with_success_task[
target_factor_task_information
] = []
for component_node in component_analysis_result:
# Reverse iterate, a trade-off with intersection search
count = 0
for task_des_node in self.knowledgebase.graph_query_by_node(
node=component_node,
step=1,
constraint_labels=["task_description"],
block=True,
)[::-1]:
if task_des_node not in task_des_node_list:
task_des_node_list.append(task_des_node)
count += 1
if count >= single_component_constraint:
break
for node in task_des_node_list:
for searched_node in self.knowledgebase.graph_query_by_node(
node=node,
step=50,
constraint_labels=[
"task_success_implement",
],
block=True,
):
if searched_node.label == "task_success_implement":
target_knowledge = self.knowledgebase.node_to_implementation_knowledge_dict[
searched_node.id
]
if (
target_knowledge
not in factor_implementation_queried_graph_knowledge.component_with_success_task[
target_factor_task_information
]
):
factor_implementation_queried_graph_knowledge.component_with_success_task[
target_factor_task_information
].append(target_knowledge)
# finally add embedding related knowledge
knowledge_base_success_task_list = list(self.knowledgebase.success_task_to_knowledge_dict)
similarity = calculate_embedding_distance_between_str_list(
[target_factor_task_information],
knowledge_base_success_task_list,
)[0]
similar_indexes = sorted(
range(len(similarity)),
key=lambda i: similarity[i],
reverse=True,
)
embedding_similar_successful_knowledge = [
self.knowledgebase.success_task_to_knowledge_dict[knowledge_base_success_task_list[index]]
for index in similar_indexes
]
for knowledge in embedding_similar_successful_knowledge:
if (
knowledge
not in factor_implementation_queried_graph_knowledge.component_with_success_task[
target_factor_task_information
]
):
factor_implementation_queried_graph_knowledge.component_with_success_task[
target_factor_task_information
].append(knowledge)
if knowledge_sampler > 0:
factor_implementation_queried_graph_knowledge.component_with_success_task[
target_factor_task_information
] = [
knowledge
for knowledge in factor_implementation_queried_graph_knowledge.component_with_success_task[
target_factor_task_information
]
if random.uniform(0, 1) <= knowledge_sampler
]
# Make sure no less than half of the knowledge are from GT
queried_knowledge_list = factor_implementation_queried_graph_knowledge.component_with_success_task[
target_factor_task_information
]
queried_from_gt_knowledge_list = [
knowledge
for knowledge in queried_knowledge_list
if knowledge.feedback is not None and knowledge.feedback.final_decision_based_on_gt == True
]
queried_without_gt_knowledge_list = [
knowledge
for knowledge in queried_knowledge_list
if knowledge.feedback is not None and knowledge.feedback.final_decision_based_on_gt == False
]
queried_from_gt_knowledge_count = max(
min(v2_query_component_limit // 2, len(queried_from_gt_knowledge_list)),
v2_query_component_limit - len(queried_without_gt_knowledge_list),
)
factor_implementation_queried_graph_knowledge.component_with_success_task[
target_factor_task_information
] = (
queried_from_gt_knowledge_list[:queried_from_gt_knowledge_count]
+ queried_without_gt_knowledge_list[: v2_query_component_limit - queried_from_gt_knowledge_count]
)
return factor_implementation_queried_graph_knowledge
def error_query(
self,
evo: EvolvableSubjects,
factor_implementation_queried_graph_knowledge: FactorImplementationQueriedGraphKnowledge,
v2_query_error_limit: int = 5,
knowledge_sampler: float = 1.0,
) -> QueriedKnowledge | None:
# queried_error_knowledge = FactorImplementationQueriedGraphErrorKnowledge()
for task_index, target_factor_task in enumerate(evo.target_factor_tasks):
target_factor_task_information = target_factor_task.get_factor_information()
factor_implementation_queried_graph_knowledge.error_with_success_task[target_factor_task_information] = {}
if (
target_factor_task_information in self.knowledgebase.success_task_to_knowledge_dict
or target_factor_task_information in factor_implementation_queried_graph_knowledge.failed_task_info_set
):
factor_implementation_queried_graph_knowledge.error_with_success_task[
target_factor_task_information
] = []
else:
factor_implementation_queried_graph_knowledge.error_with_success_task[
target_factor_task_information
] = []
if (
target_factor_task_information in self.knowledgebase.working_trace_error_analysis
and len(self.knowledgebase.working_trace_error_analysis[target_factor_task_information]) > 0
and len(factor_implementation_queried_graph_knowledge.former_traces[target_factor_task_information])
> 0
):
queried_last_trace = factor_implementation_queried_graph_knowledge.former_traces[
target_factor_task_information
][-1]
target_index = self.knowledgebase.working_trace_knowledge[target_factor_task_information].index(
queried_last_trace,
)
last_knowledge_error_analysis_result = self.knowledgebase.working_trace_error_analysis[
target_factor_task_information
][target_index]
else:
last_knowledge_error_analysis_result = []
error_nodes = []
for error_node in last_knowledge_error_analysis_result:
if not isinstance(error_node, UndirectedNode):
error_node = self.knowledgebase.graph_get_node_by_content(content=error_node)
if error_node is None:
continue
error_nodes.append(error_node)
if len(error_nodes) > 1:
task_trace_node_list = self.knowledgebase.graph_query_by_intersection(
error_nodes,
constraint_labels=["task_trace"],
output_intersection_origin=True,
)
single_error_constraint = (v2_query_error_limit // len(error_nodes)) + 1
else:
task_trace_node_list = []
single_error_constraint = v2_query_error_limit
for error_node in error_nodes:
# Reverse iterate, a trade-off with intersection search
count = 0
for task_trace_node in self.knowledgebase.graph_query_by_node(
node=error_node,
step=1,
constraint_labels=["task_trace"],
block=True,
)[::-1]:
if task_trace_node not in task_trace_node_list:
task_trace_node_list.append([[error_node], task_trace_node])
count += 1
if count >= single_error_constraint:
break
# for error_node in last_knowledge_error_analysis_result:
# if not isinstance(error_node, UndirectedNode):
# error_node = self.knowledgebase.graph_get_node_by_content(content=error_node)
# if error_node is None:
# continue
# for searched_node in self.knowledgebase.graph_query_by_node(
# node=error_node,
# step=1,
# constraint_labels=["task_trace"],
# block=True,
# ):
# if searched_node not in [node[0] for node in task_trace_node_list]:
# task_trace_node_list.append((searched_node, error_node.content))
same_error_success_knowledge_pair_list = []
same_error_success_node_set = set()
for error_node_list, trace_node in task_trace_node_list:
for searched_trace_success_node in self.knowledgebase.graph_query_by_node(
node=trace_node,
step=50,
constraint_labels=[
"task_trace",
"task_success_implement",
"task_description",
],
block=True,
):
if (
searched_trace_success_node not in same_error_success_node_set
and searched_trace_success_node.label == "task_success_implement"
):
same_error_success_node_set.add(searched_trace_success_node)
trace_knowledge = self.knowledgebase.node_to_implementation_knowledge_dict[trace_node.id]
success_knowledge = self.knowledgebase.node_to_implementation_knowledge_dict[
searched_trace_success_node.id
]
error_content = ""
for index, error_node in enumerate(error_node_list):
error_content += f"{index+1}. {error_node.content}; "
same_error_success_knowledge_pair_list.append(
(
error_content,
(trace_knowledge, success_knowledge),
),
)
if knowledge_sampler > 0:
same_error_success_knowledge_pair_list = [
knowledge
for knowledge in same_error_success_knowledge_pair_list
if random.uniform(0, 1) <= knowledge_sampler
]
same_error_success_knowledge_pair_list = same_error_success_knowledge_pair_list[:v2_query_error_limit]
factor_implementation_queried_graph_knowledge.error_with_success_task[
target_factor_task_information
] = same_error_success_knowledge_pair_list
return factor_implementation_queried_graph_knowledge
class FactorImplementationGraphKnowledgeBase(KnowledgeBase):
def __init__(self, init_component_list=None) -> None:
"""
Load knowledge, offer brief information of knowledge and common handle interfaces
"""
self.graph: UndirectedGraph = UndirectedGraph.load(Path.cwd() / "graph.pkl")
FinCoLog().info(f"Knowledge Graph loaded, size={self.graph.size()}")
if init_component_list:
for component in init_component_list:
exist_node = self.graph.get_node_by_content(content=component)
node = exist_node if exist_node else UndirectedNode(content=component, label="component")
self.graph.add_nodes(node=node, neighbors=[])
# A dict containing all working trace until they fail or succeed
self.working_trace_knowledge = {}
# A dict containing error analysis each step aligned with working trace
self.working_trace_error_analysis = {}
# Add already success task
self.success_task_to_knowledge_dict = {}
# key:node_id(for task trace and success implement), value:knowledge instance(aka 'FactorImplementationKnowledge')
self.node_to_implementation_knowledge_dict = {}
# store the task description to component nodes
self.task_to_component_nodes = {}
def get_all_nodes_by_label(self, label: str) -> list[UndirectedNode]:
return self.graph.get_all_nodes_by_label(label)
def update_success_task(
self,
success_task_info: str,
): # Transfer the success tasks' working trace to knowledge storage & graph
success_task_trace = self.working_trace_knowledge[success_task_info]
success_task_error_analysis_record = (
self.working_trace_error_analysis[success_task_info]
if success_task_info in self.working_trace_error_analysis
else []
)
task_des_node = UndirectedNode(content=success_task_info, label="task_description")
self.graph.add_nodes(
node=task_des_node,
neighbors=self.task_to_component_nodes[success_task_info],
) # 1st version, we assume that all component nodes are given
for index, trace_unit in enumerate(success_task_trace): # every unit: single_knowledge
neighbor_nodes = [task_des_node]
if index != len(success_task_trace) - 1:
trace_node = UndirectedNode(
content=trace_unit.get_implementation_and_feedback_str(),
label="task_trace",
)
self.node_to_implementation_knowledge_dict[trace_node.id] = trace_unit
for node_index, error_node in enumerate(success_task_error_analysis_record[index]):
if type(error_node).__name__ == "str":
queried_node = self.graph.get_node_by_content(content=error_node)
if queried_node is None:
new_error_node = UndirectedNode(content=error_node, label="error")
self.graph.add_node(node=new_error_node)
success_task_error_analysis_record[index][node_index] = new_error_node
else:
success_task_error_analysis_record[index][node_index] = queried_node
neighbor_nodes.extend(success_task_error_analysis_record[index])
self.graph.add_nodes(node=trace_node, neighbors=neighbor_nodes)
else:
success_node = UndirectedNode(
content=trace_unit.get_implementation_and_feedback_str(),
label="task_success_implement",
)
self.graph.add_nodes(node=success_node, neighbors=neighbor_nodes)
self.node_to_implementation_knowledge_dict[success_node.id] = trace_unit
def query(self):
pass
def graph_get_node_by_content(self, content: str) -> UndirectedNode:
return self.graph.get_node_by_content(content=content)
def graph_query_by_content(
self,
content: Union[str, list[str]],
topk_k: int = 5,
step: int = 1,
constraint_labels: list[str] = None,
constraint_node: UndirectedNode = None,
similarity_threshold: float = 0.0,
constraint_distance: float = 0,
block: bool = False,
) -> list[UndirectedNode]:
"""
search graph by content similarity and connection relationship, return empty list if nodes' chain without node
near to constraint_node
Parameters
----------
constraint_distance
content
topk_k: the upper number of output for each query, if the number of fit nodes is less than topk_k, return all fit nodes's content
step
constraint_labels
constraint_node
similarity_threshold
block: despite the start node, the search can only flow through the constraint_label type nodes
Returns
-------
"""
return self.graph.query_by_content(
content=content,
topk_k=topk_k,
step=step,
constraint_labels=constraint_labels,
constraint_node=constraint_node,
similarity_threshold=similarity_threshold,
constraint_distance=constraint_distance,
block=block,
)
def graph_query_by_node(
self,
node: UndirectedNode,
step: int = 1,
constraint_labels: list[str] = None,
constraint_node: UndirectedNode = None,
constraint_distance: float = 0,
block: bool = False,
) -> list[UndirectedNode]:
"""
search graph by connection, return empty list if nodes' chain without node near to constraint_node
Parameters
----------
node : start node
step : the max steps will be searched
constraint_labels : the labels of output nodes
constraint_node : the node that the output nodes must connect to
constraint_distance : the max distance between output nodes and constraint_node
block: despite the start node, the search can only flow through the constraint_label type nodes
Returns
-------
A list of nodes
"""
nodes = self.graph.query_by_node(
node=node,
step=step,
constraint_labels=constraint_labels,
constraint_node=constraint_node,
constraint_distance=constraint_distance,
block=block,
)
return nodes
def graph_query_by_intersection(
self,
nodes: list[UndirectedNode],
steps: int = 1,
constraint_labels: list[str] = None,
output_intersection_origin: bool = False,
) -> list[UndirectedNode] | list[list[list[UndirectedNode], UndirectedNode]]:
"""
search graph by node intersection, node intersected by a higher frequency has a prior order in the list
Parameters
----------
nodes : node list
step : the max steps will be searched
constraint_labels : the labels of output nodes
output_intersection_origin: output the list that contains the node which form this intersection node
Returns
-------
A list of nodes
"""
node_count = len(nodes)
assert node_count >= 2, "nodes length must >=2"
intersection_node_list = []
if output_intersection_origin:
origin_list = []
for k in range(node_count, 1, -1):
possible_combinations = combinations(nodes, k)
for possible_combination in possible_combinations:
node_list = list(possible_combination)
intersection_node_list.extend(
self.graph.get_nodes_intersection(node_list, steps=steps, constraint_labels=constraint_labels)
)
if output_intersection_origin:
for _ in range(len(intersection_node_list)):
origin_list.append(node_list)
intersection_node_list_sort_by_freq = []
for index, node in enumerate(intersection_node_list):
if node not in intersection_node_list_sort_by_freq:
if output_intersection_origin:
intersection_node_list_sort_by_freq.append([origin_list[index], node])
else:
intersection_node_list_sort_by_freq.append(node)
return intersection_node_list_sort_by_freq
@@ -0,0 +1,42 @@
from pathlib import Path
from finco.conf import FincoSettings
class FactorImplementSettings(FincoSettings):
file_based_execution_data_folder: str = str(
(Path().cwd() / "git_ignore_folder" / "factor_implementation_source_data").absolute(),
)
file_based_execution_workspace: str = str(
(Path().cwd() / "git_ignore_folder" / "factor_implementation_workspace").absolute(),
)
implementation_execution_cache_location: str = str(
(Path().cwd() / "git_ignore_folder" / "factor_implementation_execution_cache.pkl").absolute(),
)
enable_execution_cache: bool = True # whether to enable the execution cache
# TODO: the factor implement specific settings should not appear in this settings
# Evolving should have a method specific settings
# evolving related config
fail_task_trial_limit: int = 20
v1_query_former_trace_limit: int = 5
v1_query_similar_success_limit: int = 5
v2_query_component_limit: int = 1
v2_query_error_limit: int = 1
v2_query_former_trace_limit: int = 1
v2_error_summary: bool = False
v2_knowledge_sampler: float = 1.0
chat_token_limit: int = (
100000 # 100000 is the maximum limit of gpt4, which might increase in the future version of gpt
)
implementation_factors_per_round: int = 100 # how many factors to choose for each round of evolving
evo_multi_proc_n: int = 16 # how many processes to use for evolving (including eval & generation)
file_based_execution_timeout: int = 120 # seconds for each factor implementation execution
FIS = FactorImplementSettings()
@@ -0,0 +1,535 @@
import json
from abc import ABC, abstractmethod
from typing import Tuple
import pandas as pd
from jinja2 import Template
from oai.llm_utils import APIBackend
from finco.log import FinCoLog
from factor_implementation.share_modules.conf import FactorImplementSettings
from factor_implementation.share_modules.factor import (
FactorImplementation,
FactorImplementationTask,
)
from factor_implementation.share_modules.prompt import (
FactorImplementationPrompts,
)
class Evaluator(ABC):
@abstractmethod
def evaluate(
self,
target_task: FactorImplementationTask,
implementation: FactorImplementation,
gt_implementation: FactorImplementation,
**kwargs,
):
raise NotImplementedError
class FactorImplementationCodeEvaluator(Evaluator):
def evaluate(
self,
target_task: FactorImplementationTask,
implementation: FactorImplementation,
execution_feedback: str,
factor_value_feedback: str = "",
gt_implementation: FactorImplementation = None,
**kwargs,
):
factor_information = target_task.get_factor_information()
code = implementation.code
system_prompt = FactorImplementationPrompts()["evaluator_code_feedback_v1_system"]
execution_feedback_to_render = execution_feedback
user_prompt = Template(
FactorImplementationPrompts()["evaluator_code_feedback_v1_user"],
).render(
factor_information=factor_information,
code=code,
execution_feedback=execution_feedback_to_render,
factor_value_feedback=factor_value_feedback,
gt_code=gt_implementation.code if gt_implementation else None,
)
while (
APIBackend().build_messages_and_calculate_token(
user_prompt=user_prompt,
system_prompt=system_prompt,
former_messages=[],
)
> FactorImplementSettings().chat_token_limit
):
execution_feedback_to_render = execution_feedback_to_render[len(execution_feedback_to_render) // 2 :]
user_prompt = Template(
FactorImplementationPrompts()["evaluator_code_feedback_v1_user"],
).render(
factor_information=factor_information,
code=code,
execution_feedback=execution_feedback_to_render,
factor_value_feedback=factor_value_feedback,
gt_code=gt_implementation.code if gt_implementation else None,
)
critic_response = APIBackend().build_messages_and_create_chat_completion(
user_prompt=user_prompt,
system_prompt=system_prompt,
json_mode=False,
)
# critic_response = json.loads(critic_response)
return critic_response
class FactorImplementationEvaluator(Evaluator):
# TODO:
# I think we should have unified interface for all evaluates, for examples.
# So we should adjust the interface of other factors
@abstractmethod
def evaluate(
self,
gt: FactorImplementation,
gen: FactorImplementation,
) -> Tuple[str, object]:
"""You can get the dataframe by
.. code-block:: python
_, gt_df = gt.execute()
_, gen_df = gen.execute()
Returns
-------
Tuple[str, object]
- str: the text-based description of the evaluation result
- object: a comparable metric (bool, integer, float ...)
"""
raise NotImplementedError("Please implement the `evaluator` method")
def _get_df(self, gt: FactorImplementation, gen: FactorImplementation):
_, gt_df = gt.execute()
_, gen_df = gen.execute()
if isinstance(gen_df, pd.Series):
gen_df = gen_df.to_frame("source_factor")
if isinstance(gt_df, pd.Series):
gt_df = gt_df.to_frame("gt_factor")
return gt_df, gen_df
# NOTE: the following evaluators are splited from FactorImplementationValueEvaluator
class FactorImplementationSingleColumnEvaluator(FactorImplementationEvaluator):
def evaluate(
self,
gt: FactorImplementation,
gen: FactorImplementation,
) -> Tuple[str, object]:
gt_df, gen_df = self._get_df(gt, gen)
if len(gen_df.columns) == 1 and len(gt_df.columns) == 1:
return "Both dataframes have only one column.", True
elif len(gen_df.columns) != 1:
gen_df = gen_df.iloc(axis=1)[
[
0,
]
]
return (
"The source dataframe has more than one column. Please check the implementation. We only evaluate the first column.",
False,
)
return "", False
def __str__(self) -> str:
return self.__class__.__name__
class FactorImplementationIndexFormatEvaluator(FactorImplementationEvaluator):
def evaluate(
self,
gt: FactorImplementation,
gen: FactorImplementation,
) -> Tuple[str, object]:
gt_df, gen_df = self._get_df(gt, gen)
idx_name_right = gen_df.index.names == ("datetime", "instrument")
if idx_name_right:
return (
'The index of the dataframe is ("datetime", "instrument") and align with the predefined format.',
True,
)
else:
return (
'The index of the dataframe is not ("datetime", "instrument"). Please check the implementation.',
False,
)
def __str__(self) -> str:
return self.__class__.__name__
class FactorImplementationRowCountEvaluator(FactorImplementationEvaluator):
def evaluate(
self,
gt: FactorImplementation,
gen: FactorImplementation,
) -> Tuple[str, object]:
gt_df, gen_df = self._get_df(gt, gen)
if gen_df.shape[0] == gt_df.shape[0]:
return "Both dataframes have the same rows count.", True
else:
return (
f"The source dataframe and the ground truth dataframe have different rows count. The source dataframe has {gen_df.shape[0]} rows, while the ground truth dataframe has {gt_df.shape[0]} rows. Please check the implementation.",
False,
)
def __str__(self) -> str:
return self.__class__.__name__
class FactorImplementationIndexEvaluator(FactorImplementationEvaluator):
def evaluate(
self,
gt: FactorImplementation,
gen: FactorImplementation,
) -> Tuple[str, object]:
gt_df, gen_df = self._get_df(gt, gen)
if gen_df.index.equals(gt_df.index):
return "Both dataframes have the same index.", True
else:
return (
"The source dataframe and the ground truth dataframe have different index. Please check the implementation.",
False,
)
def __str__(self) -> str:
return self.__class__.__name__
class FactorImplementationMissingValuesEvaluator(FactorImplementationEvaluator):
def evaluate(
self,
gt: FactorImplementation,
gen: FactorImplementation,
) -> Tuple[str, object]:
gt_df, gen_df = self._get_df(gt, gen)
if gen_df.isna().sum().sum() == gt_df.isna().sum().sum():
return "Both dataframes have the same missing values.", True
else:
return (
f"The dataframes do not have the same missing values. The source dataframe has {gen_df.isna().sum().sum()} missing values, while the ground truth dataframe has {gt_df.isna().sum().sum()} missing values. Please check the implementation.",
False,
)
def __str__(self) -> str:
return self.__class__.__name__
class FactorImplementationValuesEvaluator(FactorImplementationEvaluator):
def evaluate(
self,
gt: FactorImplementation,
gen: FactorImplementation,
) -> Tuple[str, object]:
gt_df, gen_df = self._get_df(gt, gen)
try:
close_values = gen_df.sub(gt_df).abs().lt(1e-6)
result_int = close_values.astype(int)
pos_num = result_int.sum().sum()
acc_rate = pos_num / close_values.size
except:
close_values = gen_df
if close_values.all().iloc[0]:
return (
"All values in the dataframes are equal within the tolerance of 1e-6.",
acc_rate,
)
else:
return (
"Some values differ by more than the tolerance of 1e-6. Check for rounding errors or differences in the calculation methods.",
acc_rate,
)
def __str__(self) -> str:
return self.__class__.__name__
class FactorImplementationCorrelationEvaluator(FactorImplementationEvaluator):
def __init__(self, hard_check: bool) -> None:
self.hard_check = hard_check
def evaluate(
self,
gt: FactorImplementation,
gen: FactorImplementation,
) -> Tuple[str, object]:
gt_df, gen_df = self._get_df(gt, gen)
concat_df = pd.concat([gen_df, gt_df], axis=1)
concat_df.columns = ["source", "gt"]
ic = concat_df.groupby("datetime").apply(lambda df: df["source"].corr(df["gt"])).dropna().mean()
ric = (
concat_df.groupby("datetime")
.apply(lambda df: df["source"].corr(df["gt"], method="spearman"))
.dropna()
.mean()
)
if self.hard_check:
if ic > 0.99 and ric > 0.99:
return (
f"The dataframes are highly correlated. The ic is {ic:.6f} and the rankic is {ric:.6f}.",
True,
)
else:
return (
f"The dataframes are not sufficiently high correlated. The ic is {ic:.6f} and the rankic is {ric:.6f}. Investigate the factors that might be causing the discrepancies and ensure that the logic of the factor calculation is consistent.",
False,
)
else:
return f"The ic is ({ic:.6f}) and the rankic is ({ric:.6f}).", ic
def __str__(self) -> str:
return self.__class__.__name__
class FactorImplementationValEvaluator(FactorImplementationEvaluator):
def evaluate(self, gt: FactorImplementation, gen: FactorImplementation):
_, gt_df = gt.execute()
_, gen_df = gen.execute()
# FIXME: refactor the two classes
fiv = FactorImplementationValueEvaluator()
return fiv.evaluate(source_df=gen_df, gt_df=gt_df)
def __str__(self) -> str:
return self.__class__.__name__
class FactorImplementationValueEvaluator(Evaluator):
# TODO: let's discuss the about the interface of the evaluator
def evaluate(
self,
source_df: pd.DataFrame,
gt_df: pd.DataFrame,
**kwargs,
) -> Tuple:
conclusions = []
if isinstance(source_df, pd.Series):
source_df = source_df.to_frame("source_factor")
conclusions.append(
"The source dataframe is a series, better convert it to a dataframe.",
)
if gt_df is not None and isinstance(gt_df, pd.Series):
gt_df = gt_df.to_frame("gt_factor")
conclusions.append(
"The ground truth dataframe is a series, convert it to a dataframe.",
)
# Check if both dataframe has only one columns
if len(source_df.columns) == 1:
conclusions.append("The source dataframe has only one column which is correct.")
else:
conclusions.append(
"The source dataframe has more than one column. Please check the implementation. We only evaluate the first column.",
)
source_df = source_df.iloc(axis=1)[
[
0,
]
]
if list(source_df.index.names) != ["datetime", "instrument"]:
conclusions.append(
rf"The index of the dataframe is not (\"datetime\", \"instrument\"), instead is {source_df.index.names}. Please check the implementation.",
)
else:
conclusions.append(
'The index of the dataframe is ("datetime", "instrument") and align with the predefined format.',
)
# Check if both dataframe have the same rows count
if gt_df is not None:
if source_df.shape[0] == gt_df.shape[0]:
conclusions.append("Both dataframes have the same rows count.")
same_row_count_result = True
else:
conclusions.append(
f"The source dataframe and the ground truth dataframe have different rows count. The source dataframe has {source_df.shape[0]} rows, while the ground truth dataframe has {gt_df.shape[0]} rows. Please check the implementation.",
)
same_row_count_result = False
# Check whether both dataframe has the same index
if source_df.index.equals(gt_df.index):
conclusions.append("Both dataframes have the same index.")
same_index_result = True
else:
conclusions.append(
"The source dataframe and the ground truth dataframe have different index. Please check the implementation.",
)
same_index_result = False
# Check for the same missing values (NaN)
if source_df.isna().sum().sum() == gt_df.isna().sum().sum():
conclusions.append("Both dataframes have the same missing values.")
same_missing_values_result = True
else:
conclusions.append(
f"The dataframes do not have the same missing values. The source dataframe has {source_df.isna().sum().sum()} missing values, while the ground truth dataframe has {gt_df.isna().sum().sum()} missing values. Please check the implementation.",
)
same_missing_values_result = False
# Check if the values are the same within a small tolerance
if not same_index_result:
conclusions.append(
"The source dataframe and the ground truth dataframe have different index. Give up comparing the values and correlation because it's useless",
)
same_values_result = False
high_correlation_result = False
else:
close_values = source_df.sub(gt_df).abs().lt(1e-6)
if close_values.all().iloc[0]:
conclusions.append(
"All values in the dataframes are equal within the tolerance of 1e-6.",
)
same_values_result = True
else:
conclusions.append(
"Some values differ by more than the tolerance of 1e-6. Check for rounding errors or differences in the calculation methods.",
)
same_values_result = False
# Check the ic and rankic between the two dataframes
concat_df = pd.concat([source_df, gt_df], axis=1)
concat_df.columns = ["source", "gt"]
try:
ic = concat_df.groupby("datetime").apply(lambda df: df["source"].corr(df["gt"])).dropna().mean()
ric = (
concat_df.groupby("datetime")
.apply(lambda df: df["source"].corr(df["gt"], method="spearman"))
.dropna()
.mean()
)
if ic > 0.99 and ric > 0.99:
conclusions.append(
f"The dataframes are highly correlated. The ic is {ic:.6f} and the rankic is {ric:.6f}.",
)
high_correlation_result = True
else:
conclusions.append(
f"The dataframes are not sufficiently high correlated. The ic is {ic:.6f} and the rankic is {ric:.6f}. Investigate the factors that might be causing the discrepancies and ensure that the logic of the factor calculation is consistent.",
)
high_correlation_result = False
# Check for shifted alignments only in the "datetime" index
max_shift_days = 2
for shift in range(-max_shift_days, max_shift_days + 1):
if shift == 0:
continue # Skip the case where there is no shift
shifted_source_df = source_df.groupby(level="instrument").shift(shift)
concat_df = pd.concat([shifted_source_df, gt_df], axis=1)
concat_df.columns = ["source", "gt"]
shifted_ric = (
concat_df.groupby("datetime")
.apply(lambda df: df["source"].corr(df["gt"], method="spearman"))
.dropna()
.mean()
)
if shifted_ric > 0.99:
conclusions.append(
f"The dataframes are highly correlated with a shift of {max_shift_days} days in the 'date' index. Shifted rankic: {shifted_ric:.6f}.",
)
break
else:
conclusions.append(
f"No sufficient correlation found when shifting up to {max_shift_days} days in the 'date' index. Investigate the factors that might be causing discrepancies.",
)
except Exception as e:
FinCoLog().warning(f"Error occurred when calculating the correlation: {str(e)}")
conclusions.append(
f"Some error occurred when calculating the correlation. Investigate the factors that might be causing the discrepancies and ensure that the logic of the factor calculation is consistent. Error: {e}",
)
high_correlation_result = False
# Combine all conclusions into a single string
conclusion_str = "\n".join(conclusions)
final_result = (same_values_result or high_correlation_result) if gt_df is not None else False
return conclusion_str, final_result
# TODO:
def shorten_prompt(tpl: str, render_kwargs: dict, shorten_key: str, max_trail: int = 10) -> str:
"""When the prompt is too long. We have to shorten it.
But we should not truncate the prompt directly, so we should find the key we want to shorten and then shorten it.
"""
# TODO: this should replace most of code in
# - FactorImplementationFinalDecisionEvaluator.evaluate
# - FactorImplementationCodeEvaluator.evaluate
class FactorImplementationFinalDecisionEvaluator(Evaluator):
def evaluate(
self,
target_task: FactorImplementationTask,
execution_feedback: str,
value_feedback: str,
code_feedback: str,
**kwargs,
) -> Tuple:
system_prompt = FactorImplementationPrompts()["evaluator_final_decision_v1_system"]
execution_feedback_to_render = execution_feedback
user_prompt = Template(
FactorImplementationPrompts()["evaluator_final_decision_v1_user"],
).render(
factor_information=target_task.get_factor_information(),
execution_feedback=execution_feedback_to_render,
code_feedback=code_feedback,
factor_value_feedback=(
value_feedback
if value_feedback is not None
else "No Ground Truth Value provided, so no evaluation on value is performed."
),
)
while (
APIBackend().build_messages_and_calculate_token(
user_prompt=user_prompt,
system_prompt=system_prompt,
former_messages=[],
)
> FactorImplementSettings().chat_token_limit
):
execution_feedback_to_render = execution_feedback_to_render[len(execution_feedback_to_render) // 2 :]
user_prompt = Template(
FactorImplementationPrompts()["evaluator_final_decision_v1_user"],
).render(
factor_information=target_task.get_factor_information(),
execution_feedback=execution_feedback_to_render,
code_feedback=code_feedback,
factor_value_feedback=(
value_feedback
if value_feedback is not None
else "No Ground Truth Value provided, so no evaluation on value is performed."
),
)
final_evaluation_dict = json.loads(
APIBackend().build_messages_and_create_chat_completion(
user_prompt=user_prompt,
system_prompt=system_prompt,
json_mode=True,
),
)
return (
final_evaluation_dict["final_decision"],
final_evaluation_dict["final_feedback"],
)
@@ -0,0 +1,26 @@
class ImplementRunException(Exception):
"""
Exceptions raised when Implementing and running code.
- start: FactorImplementationTask => FactorGenerator
- end: Get dataframe after execution
The more detailed evaluation in dataframe values are managed by the evaluator.
"""
class CodeFormatException(ImplementRunException):
"""
The generated code is not found due format error.
"""
class RuntimeErrorException(ImplementRunException):
"""
The generated code fail to execute the script.
"""
class NoOutputException(ImplementRunException):
"""
The code fail to generate output file.
"""
@@ -0,0 +1,221 @@
import pickle
import subprocess
import uuid
from abc import ABC, abstractmethod
from pathlib import Path
from typing import Tuple, Union
import pandas as pd
from filelock import FileLock
from oai.llm_utils import md5_hash
from finco.log import FinCoLog
from factor_implementation.share_modules.conf import FactorImplementSettings
from factor_implementation.share_modules.exception import (
CodeFormatException,
NoOutputException,
RuntimeErrorException,
)
class FactorImplementationTask:
# TODO: remove the factor_ prefix may be better
def __init__(
self,
factor_name,
factor_description,
factor_formulation,
factor_formulation_description,
variables: dict = {},
) -> None:
self.factor_name = factor_name
self.factor_description = factor_description
self.factor_formulation = factor_formulation
self.factor_formulation_description = factor_formulation_description
# TODO: check variables a good candidate
self.variables = variables
def get_factor_information(self):
return f"""factor_name: {self.factor_name}
factor_description: {self.factor_description}
factor_formulation: {self.factor_formulation}
factor_formulation_description: {self.factor_formulation_description}"""
@staticmethod
def from_dict(dict):
return FactorImplementationTask(**dict)
def __repr__(self) -> str:
return f"<{self.__class__.__name__}[{self.factor_name}]>"
class FactorImplementation(ABC):
def __init__(self, target_task: FactorImplementationTask) -> None:
self.target_task = target_task
@abstractmethod
def execute(self, *args, **kwargs) -> Tuple[str, pd.DataFrame]:
raise NotImplementedError("__call__ method is not implemented.")
class FileBasedFactorImplementation(FactorImplementation):
"""
This class is used to implement a factor by writing the code to a file.
Input data and output factor value are also written to files.
"""
# TODO: (Xiao) think raising errors may get better information for processing
FB_FROM_CACHE = "The factor value has been executed and stored in the instance variable."
FB_EXEC_SUCCESS = "Execution succeeded without error."
FB_CODE_NOT_SET = "code is not set."
FB_EXECUTION_SUCCEEDED = "Execution succeeded without error."
FB_OUTPUT_FILE_NOT_FOUND = "\nExpected output file not found."
FB_OUTPUT_FILE_FOUND = "\nExpected output file found."
def __init__(
self,
target_task: FactorImplementationTask,
code,
executed_factor_value_dataframe=None,
raise_exception=False,
) -> None:
super().__init__(target_task)
self.code = code
self.executed_factor_value_dataframe = executed_factor_value_dataframe
self.logger = FinCoLog()
self.raise_exception = raise_exception
self.workspace_path = Path(
FactorImplementSettings().file_based_execution_workspace,
) / str(uuid.uuid4())
@staticmethod
def link_data_to_workspace(data_path: Path, workspace_path: Path):
data_path = Path(data_path)
workspace_path = Path(workspace_path)
for data_file_path in data_path.iterdir():
workspace_data_file_path = workspace_path / data_file_path.name
if workspace_data_file_path.exists():
workspace_data_file_path.unlink()
subprocess.run(
["ln", "-s", data_file_path, workspace_data_file_path],
check=False,
)
def execute(self, store_result: bool = False) -> Tuple[str, pd.DataFrame]:
"""
execute the implementation and get the factor value by the following steps:
1. make the directory in workspace path
2. write the code to the file in the workspace path
3. link all the source data to the workspace path folder
4. execute the code
5. read the factor value from the output file in the workspace path folder
returns the execution feedback as a string and the factor value as a pandas dataframe
parameters:
store_result: if True, store the factor value in the instance variable, this feature is to be used in the gt implementation to avoid multiple execution on the same gt implementation
"""
if self.code is None:
if self.raise_exception:
raise CodeFormatException(self.FB_CODE_NOT_SET)
else:
# TODO: to make the interface compatible with previous code. I kept the original behavior.
raise ValueError(self.FB_CODE_NOT_SET)
with FileLock(self.workspace_path / "execution.lock"):
(Path.cwd() / "git_ignore_folder" / "factor_implementation_execution_cache").mkdir(
exist_ok=True, parents=True
)
if FactorImplementSettings().enable_execution_cache:
# NOTE: cache the result for the same code
target_file_name = md5_hash(self.code)
cache_file_path = (
Path.cwd()
/ "git_ignore_folder"
/ "factor_implementation_execution_cache"
/ f"{target_file_name}.pkl"
)
if cache_file_path.exists() and not self.raise_exception:
cached_res = pickle.load(open(cache_file_path, "rb"))
if store_result and cached_res[1] is not None:
self.executed_factor_value_dataframe = cached_res[1]
return cached_res
if self.executed_factor_value_dataframe is not None:
return self.FB_FROM_CACHE, self.executed_factor_value_dataframe
source_data_path = Path(
FactorImplementSettings().file_based_execution_data_folder,
)
self.workspace_path.mkdir(exist_ok=True, parents=True)
code_path = self.workspace_path / f"{self.target_task.factor_name}.py"
code_path.write_text(self.code)
self.link_data_to_workspace(source_data_path, self.workspace_path)
execution_feedback = self.FB_EXECUTION_SUCCEEDED
try:
subprocess.check_output(
f"python {code_path}",
shell=True,
cwd=self.workspace_path,
stderr=subprocess.STDOUT,
timeout=FactorImplementSettings().file_based_execution_timeout,
)
except subprocess.CalledProcessError as e:
import site
execution_feedback = (
e.output.decode()
.replace(str(code_path.parent.absolute()), r"/path/to")
.replace(str(site.getsitepackages()[0]), r"/path/to/site-packages")
)
if len(execution_feedback) > 2000:
execution_feedback = (
execution_feedback[:1000] + "....hidden long error message...." + execution_feedback[-1000:]
)
if self.raise_exception:
raise RuntimeErrorException(execution_feedback)
except subprocess.TimeoutExpired:
execution_feedback += f"Execution timeout error and the timeout is set to {FactorImplementSettings().file_based_execution_timeout} seconds."
if self.raise_exception:
raise RuntimeErrorException(execution_feedback)
workspace_output_file_path = self.workspace_path / "result.h5"
if not workspace_output_file_path.exists():
execution_feedback += self.FB_OUTPUT_FILE_NOT_FOUND
executed_factor_value_dataframe = None
if self.raise_exception:
raise NoOutputException(execution_feedback)
else:
try:
executed_factor_value_dataframe = pd.read_hdf(workspace_output_file_path)
execution_feedback += self.FB_OUTPUT_FILE_FOUND
except Exception as e:
execution_feedback += f"Error found when reading hdf file: {e}"[:1000]
executed_factor_value_dataframe = None
if store_result and executed_factor_value_dataframe is not None:
self.executed_factor_value_dataframe = executed_factor_value_dataframe
if FactorImplementSettings().enable_execution_cache:
pickle.dump(
(execution_feedback, executed_factor_value_dataframe),
open(cache_file_path, "wb"),
)
return execution_feedback, executed_factor_value_dataframe
def __str__(self) -> str:
# NOTE:
# If the code cache works, the workspace will be None.
return f"File Factor[{self.target_task.factor_name}]: {self.workspace_path}"
def __repr__(self) -> str:
return self.__str__()
@staticmethod
def from_folder(task: FactorImplementationTask, path: Union[str, Path], **kwargs):
path = Path(path)
factor_path = (path / task.factor_name).with_suffix(".py")
with factor_path.open("r") as f:
code = f.read()
return FileBasedFactorImplementation(task, code=code, **kwargs)
@@ -0,0 +1,31 @@
from abc import ABC, abstractmethod
from typing import List
from factor_implementation.share_modules.factor import (
FactorImplementation,
FactorImplementationTask,
)
class FactorGenerator(ABC):
"""
Because implementing factors will help each other in the process of implementation, we use the interface `List[FactorImplementationTask] -> List[FactorImplementation]` instead of single factor .
"""
def __init__(self, target_task_l: List[FactorImplementationTask]) -> None:
self.target_task_l = target_task_l
@abstractmethod
def generate(self, *args, **kwargs) -> List[FactorImplementation]:
raise NotImplementedError("generate method is not implemented.")
def collect_feedback(self, feedback_obj_l: List[object]):
"""
When online evaluation.
The preivous feedbacks will be collected to support advanced factor generator
Parameters
----------
feedback_obj_l : List[object]
"""
@@ -0,0 +1,23 @@
from pathlib import Path
from typing import Dict
import yaml
from finco.utils import SingletonBaseClass
class FactorImplementationPrompts(Dict, SingletonBaseClass):
def __init__(self):
super().__init__()
prompt_yaml_path = Path(__file__).parent / "prompts.yaml"
prompt_yaml_dict = yaml.load(
open(
prompt_yaml_path,
encoding="utf8",
),
Loader=yaml.FullLoader,
)
for key, value in prompt_yaml_dict.items():
self[key] = value
@@ -0,0 +1,196 @@
evaluator_code_feedback_v1_system: |-
Your job is to give critic to user's code. User's code is expected to implement some factors in quant investment. The code contains reading data from a HDF5(H5) file, calculate the factor to each instrument on each datetime, and save the result pandas dataframe to a HDF5(H5) file.
User will firstly provide you the information of the factor, which includes the name of the factor, description of the factor, the formulation of the factor and the description of the formulation. You can check whether user's code is align with the factor.
The user will provide the source python code and the execution error message if execution failed.
The user might provide you the ground truth code for you to provide the critic. You should not leak the ground truth code to the user in any form but you can use it to provide the critic.
User has also compared the factor values calculated by the user's code and the ground truth code. The user will provide you some analyze result comparing two output. You may find some error in the code which caused the difference between the two output.
If the ground truth code is provided, your critic should only consider checking whether the user's code is align with the ground truth code since the ground truth is definitely correct.
If the ground truth code is not provided, your critic should consider checking whether the user's code is reasonable and correct.
You should provide the suggestion to each of your critic to help the user improve the code. Please response the critic in the following format. Here is an example structure for the output:
critic 1: The critic message to critic 1
critic 2: The critic message to critic 2
evaluator_code_feedback_v1_user: |-
--------------Factor information:---------------
{{ factor_information }}
--------------Python code:---------------
{{ code }}
--------------Execution feedback:---------------
{{ execution_feedback }}
{% if factor_value_feedback is not none %}
--------------Factor value feedback:---------------
{{ factor_value_feedback }}
{% endif %}
{% if gt_code is not none %}
--------------Ground truth Python code:---------------
{{ gt_code }}
{% endif %}
evolving_strategy_factor_implementation_v1_system: |-
The user is trying to implement some factors in quant investment, and you are the one to help write the python code.
{{ data_info }}
The user will provide you a formulation of the factor, which contains some function calls and some operators. You need to implement the function calls and operators in python. Your code is expected to align the formulation in any form which means The user needs to get the exact factor values with your code as expected.
Your code should contain the following part: the import part, the function part, and the main part. You should write a main function name: "calculate_{function_name}" and call this function in "if __name__ == __main__" part. Don't write any try-except block in your code. The user will catch the exception message and provide the feedback to you.
User will write your code into a python file and execute the file directly with "python {your_file_name}.py". You should calculate the factor values and save the result into a HDF5(H5) file named "result.h5" in the same directory as your python file. The result file is a HDF5(H5) file containing a pandas dataframe. The index of the dataframe is the "datetime" and "instrument", and the single column name is the factor name,and the value is the factor value. The result file should be saved in the same directory as your python file.
To help you write the correct code, the user might provide multiple information that helps you write the correct code:
1. The user might provide you the correct code to similar factors. Your should learn from these code to write the correct code.
2. The user might provide you the failed former code and the corresponding feedback to the code. The feedback contains to the execution, the code and the factor value. You should analyze the feedback and try to correct the latest code.
3. The user might provide you the suggestion to the latest fail code and some similar fail to correct pairs. Each pair contains the fail code with similar error and the corresponding corrected version code. You should learn from these suggestion to write the correct code.
Your must write your code based on your former lastest attempt below which consists of your former code and code feedback, you should read the former attempt carefully and must not modify the right part of your former code.
{% if queried_former_failed_knowledge|length != 0 %}
--------------Your former latest attempt:---------------
{% for former_failed_knowledge in queried_former_failed_knowledge %}
=====Code to implementation {{ loop.index }}=====
{{ former_failed_knowledge.implementation.code }}
=====Feedback to implementation {{ loop.index }}=====
{{ former_failed_knowledge.feedback }}
{% endfor %}
{% endif %}
A typical format of `result.h5` may be like following:
datetime instrument
2020-01-02 SZ000001 -0.001796
SZ000166 0.005780
SZ000686 0.004228
SZ000712 0.001298
SZ000728 0.005330
...
2021-12-31 SZ000750 0.000000
SZ000776 0.002459
Please response the code in the following json format. Here is an example structure for the JSON output:
{
"code": "The Python code as a string."
}
evolving_strategy_factor_implementation_v1_user: |-
--------------Target factor information:---------------
{{ factor_information_str }}
{% if queried_similar_successful_knowledge|length != 0 %}
--------------Correct code to similar factors:---------------
{% for similar_successful_knowledge in queried_similar_successful_knowledge %}
=====Factor {{loop.index}}:=====
{{ similar_successful_knowledge.target_task.get_factor_information() }}
=====Code:=====
{{ similar_successful_knowledge.implementation.code }}
{% endfor %}
{% endif %}
{% if queried_former_failed_knowledge|length != 0 %}
--------------Former failed code:---------------
{% for former_failed_knowledge in queried_former_failed_knowledge %}
=====Code to implementation {{ loop.index }}=====
{{ former_failed_knowledge.implementation.code }}
=====Feedback to implementation {{ loop.index }}=====
{{ former_failed_knowledge.feedback }}
{% endfor %}
{% endif %}
evaluator_final_decision_v1_system: |-
User is trying to implement some factor in quant investment and has finished a version of implementation. User has finished evaluation and got some feedback from the evaluator.
The evaluator run the code and get the factor value dataframe and provide several feedback regarding user's code and code output. You should analyze the feedback and considering the factor description to give a final decision about the evaluation result. The final decision concludes whether the factor is implemented correctly and if not, detail feedback containing reason and suggestion if the final decision is False.
The implementation final decision is considered in the following logic:
1. If the value and the ground truth value are exactly the same under a small tolerance, the implementation is considered correct.
2. If the value and the ground truth value have a high correlation on ic or rank ic, the implementation is considered correct.
3. If no ground truth value is not provided, the implementation is considered correct if the code execution is successful and the code feedback is reasonable.
Please response the critic in the json format. Here is an example structure for the JSON output, please strictly follow the format:
{
"final_decision": True,
"final_feedback": "The final feedback message",
}
evaluator_final_decision_v1_user: |-
--------------Factor information:---------------
{{ factor_information }}
--------------Execution feedback:---------------
{{ execution_feedback }}
--------------Code feedback:---------------
{{ code_feedback }}
--------------Factor value feedback:---------------
{{ factor_value_feedback }}
analyze_component_prompt_v1_system: |-
User is getting a new task that might consist of the components below (given in component_index: component_description):
{{all_component_content}}
You should find out what components does the new task have, and put their indices in a list.
Please response the critic in the json format. Here is an example structure for the JSON output, please strictly follow the format:
{
"component_no_list": the list containing indices of components.
}
evolving_strategy_factor_implementation_v2_user: |-
--------------Target factor information:---------------
{{ factor_information_str }}
{% if queried_similar_error_knowledge|length != 0 %}
{% if not error_summary %}
Recall your last failure, your implementation met some errors.
When doing other tasks, you met some similar errors but you finally solve them. Here are some examples:
{% for error_content, similar_error_knowledge in queried_similar_error_knowledge %}
--------------Factor information to similar error ({{error_content}}):---------------
{{ similar_error_knowledge[0].target_task.get_factor_information() }}
=====Code with similar error ({{error_content}}):=====
{{ similar_error_knowledge[0].implementation.code }}
=====Success code to former code with similar error ({{error_content}}):=====
{{ similar_error_knowledge[1].implementation.code }}
{% endfor %}
{% else %}
Recall your last failure, your implementation met some errors.
After reviewing some similar errors and their solutions, here are some suggestions for you to correct your code:
{{error_summary_critics}}
{% endif %}
{% endif %}
{% if queried_similar_component_knowledge|length != 0 %}
Here are some success implements of similar component tasks, take them as references:
--------------Correct code to similar factors:---------------
{% for similar_component_knowledge in queried_similar_component_knowledge %}
=====Factor {{loop.index}}:=====
{{ similar_component_knowledge.target_task.get_factor_information() }}
=====Code:=====
{{ similar_component_knowledge.implementation.code }}
{% endfor %}
{% endif %}
evolving_strategy_error_summary_v2_system: |-
You are doing the following task:
{{factor_information_str}}
You have written some code but it meets errors like the following:
{{code_and_feedback}}
The user has found some tasks that met similar errors, and their final correct solutions.
Please refer to these similar errors and their solutions, provide some clear, short and accurate critics that might help you solve the issues in your code.
Please response the critic in the following format. Here is an example structure for the output:
critic 1: The critic message to critic 1
critic 2: The critic message to critic 2
evolving_strategy_error_summary_v2_user: |-
{% if queried_similar_error_knowledge|length != 0 %}
{% for error_content, similar_error_knowledge in queried_similar_error_knowledge %}
--------------Factor information to similar error ({{error_content}}):---------------
{{ similar_error_knowledge[0].target_task.get_factor_information() }}
=====Code with similar error ({{error_content}}):=====
{{ similar_error_knowledge[0].implementation.code }}
=====Success code to former code with similar error ({{error_content}}):=====
{{ similar_error_knowledge[1].implementation.code }}
{% endfor %}
{% endif %}
@@ -0,0 +1,51 @@
from pathlib import Path
import pandas as pd
# render it with jinja
from jinja2 import Template
from factor_implementation.share_modules.conf import FIS
TPL = """
{{file_name}}
```{{type_desc}}
{{content}}
````
"""
# Create a Jinja template from the string
JJ_TPL = Template(TPL)
def get_data_folder_intro():
"""Direclty get the info of the data folder.
It is for preparing prompting message.
"""
content_l = []
for p in Path(FIS.file_based_execution_data_folder).iterdir():
if p.name.endswith(".h5"):
df = pd.read_hdf(p)
# get df.head() as string with full width
pd.set_option("display.max_columns", None) # or 1000
pd.set_option("display.max_rows", None) # or 1000
pd.set_option("display.max_colwidth", None) # or 199
rendered = JJ_TPL.render(
file_name=p.name,
type_desc="generated by `pd.read_hdf(filename).head()`",
content=df.head().to_string(),
)
content_l.append(rendered)
elif p.name.endswith(".md"):
with open(p) as f:
content = f.read()
rendered = JJ_TPL.render(
file_name=p.name,
type_desc="markdown",
content=content,
)
content_l.append(rendered)
else:
raise NotImplementedError(
f"file type {p.name} is not supported. Please implement its description function.",
)
return "\n ----------------- file spliter -------------\n".join(content_l)