mirror of
https://github.com/NicolasBohn/NexQuant.git
synced 2026-08-05 11:07:43 +00:00
factor extraction pipeline ready (#16)
* run the code * update code * remove some redundant code --------- Co-authored-by: xuyang1 <xuyang1@microsoft.com>
This commit is contained in:
@@ -12,117 +12,25 @@ import tiktoken
|
||||
import yaml
|
||||
from azure.ai.formrecognizer import DocumentAnalysisClient
|
||||
from azure.core.credentials import AzureKeyCredential
|
||||
from rdagent.core.conf import FincoSettings as Config
|
||||
from rdagent.core.log import FinCoLog
|
||||
from rdagent.core.prompts import Prompts
|
||||
from jinja2 import Template
|
||||
from rdagent.oai.llm_utils import APIBackend, create_embedding_with_multiprocessing
|
||||
from sklearn.cluster import KMeans
|
||||
from sklearn.metrics.pairwise import cosine_similarity
|
||||
from sklearn.preprocessing import normalize
|
||||
|
||||
from core.conf import FincoSettings as Config
|
||||
from core.log import FinCoLog
|
||||
from oai.llm_utils import APIBackend, create_embedding_with_multiprocessing
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from langchain_core.documents import Document
|
||||
|
||||
from langchain.document_loaders import PyPDFDirectoryLoader, PyPDFLoader
|
||||
|
||||
with (Path(__file__).parent / "util_prompt.yaml").open(encoding="utf8") as f:
|
||||
UTIL_PROMPT = yaml.safe_load(
|
||||
f,
|
||||
)
|
||||
|
||||
|
||||
def load_documents_by_langchain(path: Path) -> list:
|
||||
"""Load documents from the specified path.
|
||||
|
||||
Args:
|
||||
path (str): The path to the directory or file containing the documents.
|
||||
|
||||
Returns:
|
||||
list: A list of loaded documents.
|
||||
"""
|
||||
loader = PyPDFDirectoryLoader(str(path), silent_errors=True) if path.is_dir() else PyPDFLoader(str(path))
|
||||
return loader.load()
|
||||
|
||||
|
||||
def process_documents_by_langchain(docs: list[Document]) -> dict[str, str]:
|
||||
"""Process a list of documents and group them by document name.
|
||||
|
||||
Args:
|
||||
docs (list): A list of documents.
|
||||
|
||||
Returns:
|
||||
dict: A dictionary where the keys are document names and the values are
|
||||
the concatenated content of the documents.
|
||||
"""
|
||||
content_dict = {}
|
||||
|
||||
for doc in docs:
|
||||
doc_name = str(Path(doc.metadata["source"]).resolve())
|
||||
doc_content = doc.page_content
|
||||
|
||||
if doc_name not in content_dict:
|
||||
content_dict[str(doc_name)] = doc_content
|
||||
else:
|
||||
content_dict[str(doc_name)] += doc_content
|
||||
|
||||
return content_dict
|
||||
|
||||
|
||||
def load_and_process_pdfs_by_langchain(path: Path) -> dict[str, str]:
|
||||
return process_documents_by_langchain(load_documents_by_langchain(path))
|
||||
|
||||
|
||||
def load_and_process_one_pdf_by_azure_document_intelligence(
|
||||
path: Path,
|
||||
key: str,
|
||||
endpoint: str,
|
||||
) -> str:
|
||||
pages = len(PyPDFLoader(str(path)).load())
|
||||
document_analysis_client = DocumentAnalysisClient(
|
||||
endpoint=endpoint,
|
||||
credential=AzureKeyCredential(key),
|
||||
)
|
||||
|
||||
with path.open("rb") as file:
|
||||
result = document_analysis_client.begin_analyze_document(
|
||||
"prebuilt-document",
|
||||
file,
|
||||
pages=f"1-{pages}",
|
||||
).result()
|
||||
return result.content
|
||||
|
||||
|
||||
def load_and_process_pdfs_by_azure_document_intelligence(path: Path) -> dict[str, str]:
|
||||
config = Config()
|
||||
|
||||
assert config.azure_document_intelligence_key is not None
|
||||
assert config.azure_document_intelligence_endpoint is not None
|
||||
|
||||
content_dict = {}
|
||||
ab_path = path.resolve()
|
||||
if ab_path.is_file():
|
||||
assert ".pdf" in ab_path.suffixes, "The file must be a PDF file."
|
||||
proc = load_and_process_one_pdf_by_azure_document_intelligence
|
||||
content_dict[str(ab_path)] = proc(
|
||||
ab_path,
|
||||
config.azure_document_intelligence_key,
|
||||
config.azure_document_intelligence_endpoint,
|
||||
)
|
||||
else:
|
||||
for file_path in ab_path.rglob("*"):
|
||||
if file_path.is_file() and ".pdf" in file_path.suffixes:
|
||||
content_dict[str(file_path)] = load_and_process_one_pdf_by_azure_document_intelligence(
|
||||
file_path,
|
||||
config.azure_document_intelligence_key,
|
||||
config.azure_document_intelligence_endpoint,
|
||||
)
|
||||
return content_dict
|
||||
document_process_prompts = Prompts(file_path=Path(__file__).parent / "prompts.yaml")
|
||||
|
||||
|
||||
def classify_report_from_dict(
|
||||
report_dict: Mapping[str, str],
|
||||
api: APIBackend,
|
||||
input_max_token: int = 128000,
|
||||
vote_time: int = 1,
|
||||
substrings: tuple[str] = (),
|
||||
@@ -132,7 +40,6 @@ def classify_report_from_dict(
|
||||
- report_dict (Dict[str, str]):
|
||||
A dictionary where the key is the path of the report (ending with .pdf),
|
||||
and the value is either the report content as a string.
|
||||
- api (APIBackend): An instance of the APIBackend class.
|
||||
- input_max_token (int): Specifying the maximum number of input tokens.
|
||||
- vote_time (int): An integer specifying how many times to vote.
|
||||
- substrings (list(str)): List of hardcode substrings.
|
||||
@@ -155,7 +62,7 @@ def classify_report_from_dict(
|
||||
)
|
||||
|
||||
res_dict = {}
|
||||
classify_prompt = UTIL_PROMPT["classify_system"]
|
||||
classify_prompt = document_process_prompts["classify_system"]
|
||||
enc = tiktoken.encoding_for_model("gpt-4-turbo")
|
||||
|
||||
for key, value in report_dict.items():
|
||||
@@ -183,7 +90,7 @@ def classify_report_from_dict(
|
||||
for _ in range(vote_time):
|
||||
user_prompt = content
|
||||
system_prompt = classify_prompt
|
||||
res = api.build_messages_and_create_chat_completion(
|
||||
res = APIBackend().build_messages_and_create_chat_completion(
|
||||
user_prompt=user_prompt,
|
||||
system_prompt=system_prompt,
|
||||
json_mode=True,
|
||||
@@ -209,7 +116,7 @@ def __extract_factors_name_and_desc_from_content(
|
||||
content: str,
|
||||
) -> dict[str, dict[str, str]]:
|
||||
session = APIBackend().build_chat_session(
|
||||
session_system_prompt=UTIL_PROMPT["extract_factors_system"],
|
||||
session_system_prompt=document_process_prompts["extract_factors_system"],
|
||||
)
|
||||
|
||||
extracted_factor_dict = {}
|
||||
@@ -228,16 +135,14 @@ def __extract_factors_name_and_desc_from_content(
|
||||
except json.JSONDecodeError:
|
||||
parse_success = False
|
||||
if ret_json_str is None or not parse_success:
|
||||
current_user_prompt = (
|
||||
"Your response didn't follow the instruction might be wrong json format. Try again."
|
||||
)
|
||||
current_user_prompt = "Your response didn't follow the instruction might be wrong json format. Try again."
|
||||
else:
|
||||
factors = ret_dict["factors"]
|
||||
if len(factors) == 0:
|
||||
break
|
||||
for factor_name, factor_description in factors.items():
|
||||
extracted_factor_dict[factor_name] = factor_description
|
||||
current_user_prompt = UTIL_PROMPT["extract_factors_follow_user"]
|
||||
current_user_prompt = document_process_prompts["extract_factors_follow_user"]
|
||||
|
||||
return extracted_factor_dict
|
||||
|
||||
@@ -251,9 +156,9 @@ def __extract_factors_formulation_from_content(
|
||||
columns=["factor_name", "factor_description"],
|
||||
)
|
||||
|
||||
system_prompt = UTIL_PROMPT["extract_factor_formulation_system"]
|
||||
system_prompt = document_process_prompts["extract_factor_formulation_system"]
|
||||
current_user_prompt = Template(
|
||||
UTIL_PROMPT["extract_factor_formulation_user"],
|
||||
document_process_prompts["extract_factor_formulation_user"],
|
||||
).render(report_content=content, factor_dict=factor_dict_df.to_string())
|
||||
|
||||
session = APIBackend().build_chat_session(session_system_prompt=system_prompt)
|
||||
@@ -272,9 +177,7 @@ def __extract_factors_formulation_from_content(
|
||||
except json.JSONDecodeError:
|
||||
parse_success = False
|
||||
if ret_json_str is None or not parse_success:
|
||||
current_user_prompt = (
|
||||
"Your response didn't follow the instruction might be wrong json format. Try again."
|
||||
)
|
||||
current_user_prompt = "Your response didn't follow the instruction might be wrong json format. Try again."
|
||||
else:
|
||||
for name, formulation_and_description in ret_dict.items():
|
||||
if name in factor_dict:
|
||||
@@ -293,7 +196,7 @@ def __extract_factors_formulation_from_content(
|
||||
return factor_to_formulation
|
||||
|
||||
|
||||
def extract_factor_and_formulation_from_one_report(
|
||||
def __extract_factor_and_formulation_from_one_report(
|
||||
content: str,
|
||||
) -> dict[str, dict[str, str]]:
|
||||
final_factor_dict_to_one_report = {}
|
||||
@@ -304,6 +207,9 @@ def extract_factor_and_formulation_from_one_report(
|
||||
factor_dict,
|
||||
)
|
||||
for factor_name in factor_dict:
|
||||
if factor_name not in factor_to_formulation:
|
||||
continue
|
||||
|
||||
final_factor_dict_to_one_report.setdefault(factor_name, {})
|
||||
final_factor_dict_to_one_report[factor_name]["description"] = factor_dict[factor_name]
|
||||
|
||||
@@ -323,7 +229,7 @@ def extract_factor_and_formulation_from_one_report(
|
||||
return final_factor_dict_to_one_report
|
||||
|
||||
|
||||
def extract_factors_from_report_dict_and_classify_result(
|
||||
def extract_factors_from_report_dict(
|
||||
report_dict: dict[str, str],
|
||||
useful_no_dict: dict[str, dict[str, str]],
|
||||
n_proc: int = 11,
|
||||
@@ -339,9 +245,7 @@ def extract_factors_from_report_dict_and_classify_result(
|
||||
final_report_factor_dict = {}
|
||||
# for file_name, content in useful_report_dict.items():
|
||||
# final_report_factor_dict.setdefault(file_name, {})
|
||||
# final_report_factor_dict[
|
||||
# file_name
|
||||
# ] = extract_factor_and_formulation_from_one_report(content)
|
||||
# final_report_factor_dict[file_name] = __extract_factor_and_formulation_from_one_report(content)
|
||||
|
||||
while len(final_report_factor_dict) != len(useful_report_dict):
|
||||
pool = mp.Pool(n_proc)
|
||||
@@ -353,7 +257,7 @@ def extract_factors_from_report_dict_and_classify_result(
|
||||
file_names.append(file_name)
|
||||
pool_result_list.append(
|
||||
pool.apply_async(
|
||||
extract_factor_and_formulation_from_one_report,
|
||||
__extract_factor_and_formulation_from_one_report,
|
||||
(content,),
|
||||
),
|
||||
)
|
||||
@@ -371,11 +275,32 @@ def extract_factors_from_report_dict_and_classify_result(
|
||||
return final_report_factor_dict
|
||||
|
||||
|
||||
def check_factor_dict_viability_simulate_json_mode(
|
||||
def merge_file_to_factor_dict_to_factor_dict(
|
||||
file_to_factor_dict: dict[str, dict],
|
||||
) -> dict:
|
||||
factor_dict = {}
|
||||
for file_name in file_to_factor_dict:
|
||||
for factor_name in file_to_factor_dict[file_name]:
|
||||
factor_dict.setdefault(factor_name, [])
|
||||
factor_dict[factor_name].append(file_to_factor_dict[file_name][factor_name])
|
||||
|
||||
factor_dict_simple_deduplication = {}
|
||||
for factor_name in factor_dict:
|
||||
if len(factor_dict[factor_name]) > 1:
|
||||
factor_dict_simple_deduplication[factor_name] = max(
|
||||
factor_dict[factor_name],
|
||||
key=lambda x: len(x["formulation"]),
|
||||
)
|
||||
else:
|
||||
factor_dict_simple_deduplication[factor_name] = factor_dict[factor_name][0]
|
||||
return factor_dict_simple_deduplication
|
||||
|
||||
|
||||
def __check_factor_dict_viability_simulate_json_mode(
|
||||
factor_df_string: str,
|
||||
) -> dict[str, dict[str, str]]:
|
||||
session = APIBackend().build_chat_session(
|
||||
session_system_prompt=UTIL_PROMPT["factor_viability_system"],
|
||||
session_system_prompt=document_process_prompts["factor_viability_system"],
|
||||
)
|
||||
current_user_prompt = factor_df_string
|
||||
|
||||
@@ -392,17 +317,15 @@ def check_factor_dict_viability_simulate_json_mode(
|
||||
except json.JSONDecodeError:
|
||||
parse_success = False
|
||||
if ret_json_str is None or not parse_success:
|
||||
current_user_prompt = (
|
||||
"Your response didn't follow the instruction might be wrong json format. Try again."
|
||||
)
|
||||
current_user_prompt = "Your response didn't follow the instruction might be wrong json format. Try again."
|
||||
else:
|
||||
return ret_dict
|
||||
return {}
|
||||
|
||||
|
||||
def check_factor_dict_viability(
|
||||
def check_factor_viability(
|
||||
factor_dict: dict[str, dict[str, str]],
|
||||
) -> dict[str, dict[str, str]]:
|
||||
) -> tuple[dict[str, dict[str, str]], dict[str, dict[str, str]]]:
|
||||
factor_viability_dict = {}
|
||||
|
||||
factor_df = pd.DataFrame(factor_dict).T
|
||||
@@ -417,7 +340,7 @@ def check_factor_dict_viability(
|
||||
|
||||
result_list.append(
|
||||
pool.apply_async(
|
||||
check_factor_dict_viability_simulate_json_mode,
|
||||
__check_factor_dict_viability_simulate_json_mode,
|
||||
(target_factor_df_string,),
|
||||
),
|
||||
)
|
||||
@@ -432,14 +355,20 @@ def check_factor_dict_viability(
|
||||
|
||||
factor_df = factor_df[~factor_df.index.isin(factor_viability_dict)]
|
||||
|
||||
# filtered_factor_dict = {
|
||||
# factor_name: factor_dict[factor_name]
|
||||
# for factor_name in factor_dict
|
||||
# if factor_viability_dict[factor_name]["viability"]
|
||||
# }
|
||||
|
||||
return factor_viability_dict
|
||||
|
||||
|
||||
def check_factor_duplication_simulate_json_mode(
|
||||
def __check_factor_duplication_simulate_json_mode(
|
||||
factor_df: pd.DataFrame,
|
||||
) -> list[list[str]]:
|
||||
session = APIBackend().build_chat_session(
|
||||
session_system_prompt=UTIL_PROMPT["factor_duplicate_system"],
|
||||
session_system_prompt=document_process_prompts["factor_duplicate_system"],
|
||||
)
|
||||
current_user_prompt = factor_df.to_string()
|
||||
|
||||
@@ -474,7 +403,7 @@ def check_factor_duplication_simulate_json_mode(
|
||||
return generated_duplicated_groups
|
||||
|
||||
|
||||
def kmeans_embeddings(embeddings: np.ndarray, k: int = 20) -> list[list[str]]:
|
||||
def __kmeans_embeddings(embeddings: np.ndarray, k: int = 20) -> list[list[str]]:
|
||||
x_normalized = normalize(embeddings)
|
||||
|
||||
kmeans = KMeans(
|
||||
@@ -528,7 +457,7 @@ def kmeans_embeddings(embeddings: np.ndarray, k: int = 20) -> list[list[str]]:
|
||||
)
|
||||
|
||||
|
||||
def deduplicate_factor_dict(factor_dict: dict[str, dict[str, str]]) -> list[list[str]]:
|
||||
def __deduplicate_factor_dict(factor_dict: dict[str, dict[str, str]]) -> list[list[str]]:
|
||||
factor_df = pd.DataFrame(factor_dict).T
|
||||
factor_df.index.names = ["factor_name"]
|
||||
|
||||
@@ -559,7 +488,7 @@ Factor variables: {variables}
|
||||
len(full_str_list) // Config().max_input_duplicate_factor_group,
|
||||
30,
|
||||
):
|
||||
kmeans_index_group = kmeans_embeddings(embeddings=embeddings, k=k)
|
||||
kmeans_index_group = __kmeans_embeddings(embeddings=embeddings, k=k)
|
||||
if len(kmeans_index_group[0]) < Config().max_input_duplicate_factor_group:
|
||||
target_k = k
|
||||
FinCoLog().info(f"K-means group number: {k}")
|
||||
@@ -572,7 +501,7 @@ Factor variables: {variables}
|
||||
result_list = []
|
||||
result_list = [
|
||||
pool.apply_async(
|
||||
check_factor_duplication_simulate_json_mode,
|
||||
__check_factor_duplication_simulate_json_mode,
|
||||
(factor_df.loc[factor_name_group, :],),
|
||||
)
|
||||
for factor_name_group in factor_name_groups
|
||||
@@ -593,13 +522,14 @@ Factor variables: {variables}
|
||||
return duplication_names_list
|
||||
|
||||
|
||||
def deduplicate_factors_several_times(
|
||||
def deduplicate_factors_by_llm(
|
||||
factor_dict: dict[str, dict[str, str]],
|
||||
factor_viability_dict: dict[str, dict[str, str]] = None,
|
||||
) -> list[list[str]]:
|
||||
final_duplication_names_list = []
|
||||
current_round_factor_dict = factor_dict
|
||||
for _ in range(10):
|
||||
duplication_names_list = deduplicate_factor_dict(current_round_factor_dict)
|
||||
duplication_names_list = __deduplicate_factor_dict(current_round_factor_dict)
|
||||
|
||||
new_round_names = []
|
||||
for duplication_names in duplication_names_list:
|
||||
@@ -611,5 +541,31 @@ def deduplicate_factors_several_times(
|
||||
if len(new_round_names) != 0:
|
||||
current_round_factor_dict = {factor_name: factor_dict[factor_name] for factor_name in new_round_names}
|
||||
else:
|
||||
return final_duplication_names_list
|
||||
return []
|
||||
break
|
||||
|
||||
final_duplication_names_list = sorted(final_duplication_names_list, key=lambda x: len(x), reverse=True)
|
||||
|
||||
to_replace_dict = {}
|
||||
for duplication_names in duplication_names_list:
|
||||
if factor_viability_dict is not None:
|
||||
viability_list = [factor_viability_dict[name]["viability"] for name in duplication_names]
|
||||
if True not in viability_list:
|
||||
continue
|
||||
target_factor_name = duplication_names[viability_list.index(True)]
|
||||
else:
|
||||
target_factor_name = duplication_names[0]
|
||||
for duplication_factor_name in duplication_names:
|
||||
if duplication_factor_name == target_factor_name:
|
||||
continue
|
||||
to_replace_dict[duplication_factor_name] = target_factor_name
|
||||
|
||||
llm_deduplicated_factor_dict = dict()
|
||||
added_lower_name_set = set()
|
||||
for factor_name in factor_dict:
|
||||
if factor_name not in to_replace_dict and factor_name.lower() not in added_lower_name_set:
|
||||
if factor_viability_dict is not None and not factor_viability_dict[factor_name]["viability"]:
|
||||
continue
|
||||
added_lower_name_set.add(factor_name.lower())
|
||||
llm_deduplicated_factor_dict[factor_name] = factor_dict[factor_name]
|
||||
|
||||
return llm_deduplicated_factor_dict, final_duplication_names_list
|
||||
|
||||
@@ -5,18 +5,12 @@ from pathlib import Path
|
||||
import yaml
|
||||
from azure.ai.formrecognizer import DocumentAnalysisClient
|
||||
from azure.core.credentials import AzureKeyCredential
|
||||
from finco.conf import FincoSettings as Config
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from langchain_core.documents import Document
|
||||
from rdagent.core.conf import FincoSettings as Config
|
||||
from rdagent.core.prompts import Prompts
|
||||
|
||||
from langchain_core.documents import Document
|
||||
from langchain.document_loaders import PyPDFDirectoryLoader, PyPDFLoader
|
||||
|
||||
with (Path(__file__).parent / "util_prompt.yaml").open(encoding="utf8") as f:
|
||||
UTIL_PROMPT = yaml.safe_load(
|
||||
f,
|
||||
)
|
||||
|
||||
|
||||
def load_documents_by_langchain(path: Path) -> list:
|
||||
"""Load documents from the specified path.
|
||||
|
||||
@@ -0,0 +1,182 @@
|
||||
extract_factors_system: |-
|
||||
用户会提供一篇金融工程研报,其中包括了量化因子和模型研究,请按照要求抽取以下信息:
|
||||
1. 概述这篇研报的主要研究思路;
|
||||
2. 抽取出所有的因子,并概述因子的计算过程,请注意有些因子可能存在于表格中,请不要遗漏,因子的名称请使用英文,不能包含空格,可用下划线连接,研报中可能不含有因子,若没有请返回空字典;
|
||||
3. 抽取研报里面的所有模型,并概述模型的计算过程,可以分步骤描述模型搭建或计算的过程,研报中可能不含有模型,若没有请返回空字典;
|
||||
|
||||
user will treat your factor name as key to store the factor, don't put any interaction message in the content. Just response the output without any interaction and explanation.
|
||||
All names should be in English.
|
||||
Respond with your analysis in JSON format. The JSON schema should include:
|
||||
```json
|
||||
{
|
||||
"summary": "The summary of this report",
|
||||
"factors": {
|
||||
"Name of factor 1": "Description to factor 1",
|
||||
"Name of factor 2": "Description to factor 2"
|
||||
},
|
||||
"models": {
|
||||
"Name of model 1": "Description to model 1",
|
||||
"Name of model 2": "Description to model 2"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
extract_factors_follow_user: |-
|
||||
Please continue extracting the factors. Please ignore factors appeared in former messages. If no factor is found, please return an empty dict.
|
||||
Notice: You should not miss any factor in the report! Some factors might appear several times in the report. You can repeat them to avoid missing other factors.
|
||||
Respond with your analysis in JSON format. The JSON schema should include:
|
||||
```json
|
||||
{
|
||||
"factors": {
|
||||
"Name of factor 1": "Description to factor 1",
|
||||
"Name of factor 2": "Description to factor 2"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
extract_factor_formulation_system: |-
|
||||
用户会提供一篇金融工程研报,和用户从中提取到的因子列表,请结合文章和用户提供的因子名称和因子描述,按照要求抽取:
|
||||
1. 因子的计算公式,使用latex格式,公式中的变量名称不能包含空格,可用下划线连接,公式中的因子名称与用户提供的因子名称保持一致;
|
||||
2. 因子公式中的变量和函数解释,请使用英文描述,变量名和函数名请与公式中的名称对齐
|
||||
|
||||
User has several source data:
|
||||
1. The Stock Trade Data Table containing information about stock trades, such as daily open, close, high, low, vwap prices, volume, and turnover;
|
||||
2. The Financial Data Table containing company financial statements such as the balance sheet, income statement, and cash flow statement;
|
||||
3. The Stock Fundamental Data Table containing basic information about stocks, like total shares outstanding, free float shares, industry classification, market classification, etc;
|
||||
4. The high frequency data containing price and volume of each stock containing open close high low volume vwap in each minute.
|
||||
Please try to expand the formulation to using the source data provided by user.
|
||||
|
||||
user will treat your factor name as key to store the factor, don't put any interaction message in the content. Just response the output without any interaction and explanation.
|
||||
You can extract part of the user's input factors if token is not enough. To avoid the situation that you don't respond in the valid format, don't extract more than thirty factors in one response.
|
||||
Be caution of the "\" in your formulation because In JSON, certain characters like the backslash need to be escaped with another backslash. Especially, _ and \_ are different in latex so use \_ to represent _ in latex.
|
||||
Respond with your analysis in JSON format. The JSON schema should include:
|
||||
```json
|
||||
{
|
||||
"name of factor 1": {
|
||||
"formulation": "latex formulation of factor 1",
|
||||
"variables": {
|
||||
"Name to variable or function 1": "Description to variable or function 1",
|
||||
"Name to variable or function 2": "Description to variable or function 2"
|
||||
}
|
||||
},
|
||||
"name of factor 2": {
|
||||
"formulation": "latex formulation of factor 2",
|
||||
"variables": {
|
||||
"Name to variable or function 1": "Description to variable or function 1",
|
||||
"Name to variable or function 2": "Description to variable or function 2"
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
extract_factor_formulation_user: |-
|
||||
===========================Report content:=============================
|
||||
{{ report_content }}
|
||||
===========================Factor list in dataframe=============================
|
||||
{{ factor_dict }}
|
||||
|
||||
classify_system: |-
|
||||
你是一个研报分类助手。用户会输入一篇金融研报。请按照要求回答:
|
||||
因子指能够解释资产收益率或价格等的变量;而模型则指机器学习或深度学习模型,利用因子等变量来预测价格或收益率变化。
|
||||
|
||||
请你对研报进行分类,考虑两个条件:
|
||||
1. 是金工量化领域中选股(需与择时,选基等严格区分开)方面的研报;
|
||||
2. 涉及了因子或模型的构成,或者是测试了它们的表现。
|
||||
如果研报同时满足上述两个条件,请输出1;若没有,请输出0。
|
||||
|
||||
请使用json进行回答。json key为:class
|
||||
|
||||
factor_viability_system: |-
|
||||
User has designed several factors in quant investment. Please help the user to check the viability of these factors.
|
||||
These factors are used to build a daily frequency strategy in China A-share market.
|
||||
|
||||
User will provide a pandas dataframe like table containing following information:
|
||||
1. The name of the factor;
|
||||
2. The simple description of the factor;
|
||||
3. The formulation of the factor in latex format;
|
||||
4. The description to the variables and functions in the formulation of the factor.
|
||||
|
||||
User has several source data:
|
||||
1. The Stock Trade Data Table containing information about stock trades, such as daily open, close, high, low, vwap prices, volume, and turnover;
|
||||
2. The Financial Data Table containing company financial statements such as the balance sheet, income statement, and cash flow statement;
|
||||
3. The Stock Fundamental Data Table containing basic information about stocks, like total shares outstanding, free float shares, industry classification, market classification, etc;
|
||||
4. The high frequency data containing price and volume of each stock containing open close high low volume vwap in each minute;
|
||||
5. The Consensus Expectations Factor containing the consensus expectations of the analysts about the future performance of the company.
|
||||
|
||||
|
||||
A viable factor should satisfy the following conditions:
|
||||
1. The factor should be able to be calculated in daily frequency;
|
||||
2. The factor should be able to be calculated based on each stock;
|
||||
3. The factor should be able to be calculated based on the source data provided by user.
|
||||
|
||||
You should give decision to each factor provided by the user. You should reject the factor based on very solid reason.
|
||||
Please return true to the viable factor and false to the non-viable factor.
|
||||
|
||||
Notice, you can just return part of the factors due to token limit. Your factor name should be the same as the user's factor name.
|
||||
|
||||
Please respond with your decision in JSON format. Just respond the output json string without any interaction and explanation.
|
||||
The JSON schema should include:
|
||||
```json
|
||||
{
|
||||
"Name to factor 1":
|
||||
{
|
||||
"viability": true,
|
||||
"reason": "The reason to the viability of this factor"
|
||||
},
|
||||
"Name to factor 2":
|
||||
{
|
||||
"viability": false,
|
||||
"reason": "The reason to the non-viability of this factor"
|
||||
}
|
||||
"Name to factor 3":
|
||||
{
|
||||
"viability": true,
|
||||
"reason": "The reason to the viability of this factor"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
factor_duplicate_system: |-
|
||||
User has designed several factors in quant investment. Please help the user to duplicate these factors.
|
||||
These factors are used to build a daily frequency strategy in China A-share market.
|
||||
|
||||
User will provide a pandas dataframe like table containing following information:
|
||||
1. The name of the factor;
|
||||
2. The simple description of the factor;
|
||||
3. The formulation of the factor in latex format;
|
||||
4. The description to the variables and functions in the formulation of the factor.
|
||||
|
||||
User wants to find whether there are duplicated groups. The factors in a duplicate group should satisfy the following conditions:
|
||||
1. They might differ in the name, description, formulation, or the description to the variables and functions in the formulation, some upper or lower case difference is included;
|
||||
2. They should be talking about exactly the same factor;
|
||||
3. If horizon information like 1 day, 5 days, 10 days, etc is provided, the horizon information should be the same.
|
||||
|
||||
To make your response valid, we have some very important constraint for you to follow! Listed here:
|
||||
1. You should be very confident to put duplicated factors into a group;
|
||||
2. A group should contain at least two factors;
|
||||
3. To a factor which has no duplication, don't put them into your response;
|
||||
4. To avoid merging too many similar factor, don't put more than ten factors into a group!
|
||||
You should always follow the above constraints to make your response valid.
|
||||
|
||||
Your response JSON schema should include:
|
||||
```json
|
||||
[
|
||||
[
|
||||
"factor name 1",
|
||||
"factor name 2"
|
||||
],
|
||||
[
|
||||
"factor name 5",
|
||||
"factor name 6"
|
||||
],
|
||||
[
|
||||
"factor name 7",
|
||||
"factor name 8",
|
||||
"factor name 9"
|
||||
]
|
||||
]
|
||||
```
|
||||
Your response is a list of lists. Each list represents a duplicate group containing all the factor names in this group.
|
||||
The factor names in the list should be unique and the factor names should be the same as the user's factor name.
|
||||
To avoid reaching token limit, don't respond more than fifty groups in one response. You should respond the output json string without any interaction and explanation.
|
||||
Reference in New Issue
Block a user