Files
NexQuant/rdagent/scenarios/qlib/experiment/utils.py
T

134 lines
3.9 KiB
Python
Raw Normal View History

2024-09-29 18:43:17 +08:00
import re
import shutil
from pathlib import Path
import pandas as pd
# render it with jinja
from jinja2 import Environment, StrictUndefined
from rdagent.components.coder.factor_coder.config import FACTOR_IMPLEMENT_SETTINGS
from rdagent.utils.env import QTDockerEnv
def generate_data_folder_from_qlib():
template_path = Path(__file__).parent / "factor_data_template"
qtde = QTDockerEnv()
qtde.prepare()
# Run the Qlib backtest
execute_log = qtde.run(
local_path=str(template_path),
entry=f"python generate.py",
)
assert (
Path(__file__).parent / "factor_data_template" / "daily_pv_all.h5"
).exists(), "daily_pv_all.h5 is not generated."
assert (
Path(__file__).parent / "factor_data_template" / "daily_pv_debug.h5"
).exists(), "daily_pv_debug.h5 is not generated."
Path(FACTOR_IMPLEMENT_SETTINGS.data_folder).mkdir(parents=True, exist_ok=True)
shutil.copy(
Path(__file__).parent / "factor_data_template" / "daily_pv_all.h5",
Path(FACTOR_IMPLEMENT_SETTINGS.data_folder) / "daily_pv.h5",
)
shutil.copy(
Path(__file__).parent / "factor_data_template" / "README.md",
Path(FACTOR_IMPLEMENT_SETTINGS.data_folder) / "README.md",
)
Path(FACTOR_IMPLEMENT_SETTINGS.data_folder_debug).mkdir(parents=True, exist_ok=True)
shutil.copy(
Path(__file__).parent / "factor_data_template" / "daily_pv_debug.h5",
Path(FACTOR_IMPLEMENT_SETTINGS.data_folder_debug) / "daily_pv.h5",
)
shutil.copy(
Path(__file__).parent / "factor_data_template" / "README.md",
Path(FACTOR_IMPLEMENT_SETTINGS.data_folder_debug) / "README.md",
)
2024-10-08 02:07:20 +08:00
def get_file_desc(p: Path) -> str:
"""
Get the description of a file based on its type.
Parameters
----------
p : Path
The path of the file.
Returns
-------
str
The description of the file.
"""
p = Path(p)
JJ_TPL = Environment(undefined=StrictUndefined).from_string(
"""
{{file_name}}
```{{type_desc}}
{{content}}
```
"""
)
if p.name.endswith(".h5"):
df = pd.read_hdf(p)
# get df.head() as string with full width
pd.set_option("display.max_columns", None) # or 1000
pd.set_option("display.max_rows", None) # or 1000
pd.set_option("display.max_colwidth", None) # or 199
return JJ_TPL.render(
file_name=p.name,
type_desc="generated by `pd.read_hdf(filename).head()`",
content=df.head().to_string(),
)
elif p.name.endswith(".md"):
with open(p) as f:
content = f.read()
return JJ_TPL.render(
file_name=p.name,
type_desc="markdown",
content=content,
)
else:
raise NotImplementedError(
f"file type {p.name} is not supported. Please implement its description function.",
)
2024-09-29 18:43:17 +08:00
def get_data_folder_intro(fname_reg: str = ".*", flags=0) -> str:
"""
Directly get the info of the data folder.
It is for preparing prompting message.
2024-09-29 18:43:17 +08:00
Parameters
----------
fname_reg : str
a regular expression to filter the file name.
flags: str
flags for re.match
Returns
-------
str
The description of the data folder.
"""
if (
not Path(FACTOR_IMPLEMENT_SETTINGS.data_folder).exists()
or not Path(FACTOR_IMPLEMENT_SETTINGS.data_folder_debug).exists()
):
2024-09-29 18:43:17 +08:00
# FIXME: (xiao) I think this is writing in a hard-coded way.
# get data folder intro does not imply that we are generating the data folder.
generate_data_folder_from_qlib()
content_l = []
for p in Path(FACTOR_IMPLEMENT_SETTINGS.data_folder_debug).iterdir():
2024-09-29 18:43:17 +08:00
if re.match(fname_reg, p.name, flags) is not None:
2024-10-08 02:07:20 +08:00
content_l.append(get_file_desc(p))
2024-09-29 18:43:17 +08:00
return "\n----------------- file splitter -------------\n".join(content_l)