mirror of
https://github.com/NicolasBohn/NexQuant.git
synced 2026-07-29 16:37:43 +00:00
Compare commits
394 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| bf9efe127d | |||
| 243d5259f7 | |||
| dd10ddf9fd | |||
| a619ff999b | |||
| c215e735c2 | |||
| 0c219b81e0 | |||
| 9f793d58b3 | |||
| f0deace7a7 | |||
| 682f657637 | |||
| af5898211b | |||
| 1a1e88c47c | |||
| 9890bb4b00 | |||
| 5a3a2bf2be | |||
| 77e9f186d5 | |||
| 2187081905 | |||
| 698f89e07c | |||
| 304d3fad2e | |||
| ba76063069 | |||
| d18271275e | |||
| a4986c415c | |||
| 155675d4bf | |||
| 15e3ba9004 | |||
| 671771a626 | |||
| 63ba3296dd | |||
| 805f337fc4 | |||
| bdaa376569 | |||
| de6b7f7648 | |||
| 5bb5722921 | |||
| 1a61d51fa8 | |||
| 2b7e0c4bd0 | |||
| 311c2bf977 | |||
| 50702e254e | |||
| 543fb2c1ef | |||
| 51a5247285 | |||
| a1843a9c4e | |||
| b1029cbd5a | |||
| 0611a0846a | |||
| 4974277f1c | |||
| a7c6fca8b4 | |||
| 1091d5ea29 | |||
| 63ebdb5ada | |||
| d6ce70b551 | |||
| 628deade72 | |||
| 29c245c6e3 | |||
| 2e39700eeb | |||
| 370adbfa1e | |||
| dbaf70ce57 | |||
| 3c6f9ef76d | |||
| 0109b044a0 | |||
| 5a62b47e95 | |||
| 5b47c0ab5c | |||
| 3e60749f70 | |||
| cfa5f382aa | |||
| 9ea44be5f8 | |||
| 37e9912ac0 | |||
| bc9cee49ba | |||
| 763dff1596 | |||
| fe11ee92e6 | |||
| 2c11752a54 | |||
| 7aea9a27dc | |||
| 0c97fe4ae7 | |||
| 586349af36 | |||
| 5b3f5f66cf | |||
| 69e5c3a7de | |||
| b76abe839b | |||
| c8c8e26812 | |||
| 42ea6f106b | |||
| 7742a9e759 | |||
| 8acd24a016 | |||
| 54465d6e82 | |||
| 90f6c94d3e | |||
| 7f4e5096e8 | |||
| a294bc74e0 | |||
| e71d8f6c3c | |||
| 5c7cdf298c | |||
| 8a0a0fa3c4 | |||
| 69c630e983 | |||
| ad12fccb61 | |||
| 9986b5f9ce | |||
| eb1cefb202 | |||
| 4f5cce0607 | |||
| 495063c385 | |||
| 318acb4922 | |||
| 8340ab5d09 | |||
| 22a0c7fe56 | |||
| 02534cfd88 | |||
| 9a8426f373 | |||
| 6eef7517ad | |||
| 29a14e1e22 | |||
| 6f9843a7df | |||
| aa97d5a09d | |||
| 6b7988fba1 | |||
| 93303873d0 | |||
| d0e3fc1573 | |||
| cbedbc7d54 | |||
| 03040de35f | |||
| a73f2dbc12 | |||
| 45931b1012 | |||
| 6c179442bf | |||
| 8affbf89a9 | |||
| 8968ab8c9a | |||
| c89321f5a2 | |||
| d6990a12b8 | |||
| 86e05e6d35 | |||
| fda5c281a8 | |||
| 4f2a915da5 | |||
| ff99e4dc2c | |||
| c5beaeeaf8 | |||
| 82b037cb63 | |||
| 456ed100a0 | |||
| 0798c05cb7 | |||
| 7dbffc516d | |||
| 4509d92ea7 | |||
| 19bb2e740e | |||
| 2fea82bee7 | |||
| c68b597069 | |||
| bbb9d36f77 | |||
| e44bb2dcc1 | |||
| 08fc2dcef3 | |||
| c2759a2a0a | |||
| 05092c5c5b | |||
| d9cee55063 | |||
| fe3337d6ff | |||
| 90f6fde3d4 | |||
| 0e3dcc2d8d | |||
| aedd0a0f72 | |||
| 214da4826f | |||
| f25d729a02 | |||
| 773b7ac439 | |||
| 19dd259165 | |||
| ab52e1b65c | |||
| 959f2fa076 | |||
| f06e6986c8 | |||
| b5631b9f78 | |||
| b662f45021 | |||
| 92d007e179 | |||
| c40f22361c | |||
| e22fb84fb3 | |||
| 80c5eb209e | |||
| 67114540b1 | |||
| 8517eb4a18 | |||
| a16c79e45b | |||
| ce23e1c30b | |||
| 92e7932126 | |||
| 31752bb8f0 | |||
| 7372d1011e | |||
| a918f941dd | |||
| dee11fbd9c | |||
| 0dcc990bd4 | |||
| 3f41e868b3 | |||
| fac39e5588 | |||
| fce5342ad4 | |||
| 44ada185f4 | |||
| 6818ada2cf | |||
| 11e8acc88b | |||
| 9415113e01 | |||
| d3414ea07d | |||
| 8bd2714814 | |||
| 200764d38b | |||
| c1c1342a9e | |||
| 756dfba53f | |||
| 1f0032bc50 | |||
| 9eded6faa8 | |||
| 8783781daa | |||
| da8fbcf4ee | |||
| 22ed8b522b | |||
| 7fc5e73f76 | |||
| e30768969d | |||
| bc8e9dc23a | |||
| 1474478768 | |||
| 260b563710 | |||
| 3ef0b21350 | |||
| 473b9000b9 | |||
| 00b31fa752 | |||
| cc8ae7e1f0 | |||
| 752de9dc67 | |||
| a99276d59b | |||
| 450571e86e | |||
| 7fc9e18a71 | |||
| 031d264957 | |||
| 6da6fb5ee5 | |||
| 525b1a2136 | |||
| 61813cba67 | |||
| b33be5ed5b | |||
| 6b856e2217 | |||
| f5f3e050ab | |||
| 9a43e15bf4 | |||
| 251800396d | |||
| 5f3e382248 | |||
| fa0dbb69fe | |||
| f1e4e9900c | |||
| 203be39ad6 | |||
| 60b66aa6c9 | |||
| b91471181c | |||
| a48abf4a84 | |||
| 5b14779f69 | |||
| 8ea8817e7d | |||
| fd747fe741 | |||
| b666397786 | |||
| 31381743f7 | |||
| e564b418e1 | |||
| d1031ed24d | |||
| d66c045115 | |||
| 7b3473cf30 | |||
| 1d6e97cb99 | |||
| 3d7b2b4567 | |||
| b8a699a0df | |||
| 93b41c1414 | |||
| bb91201dd2 | |||
| e0ae91b497 | |||
| b20692cbd4 | |||
| 78e2665617 | |||
| 40f2b8f98b | |||
| b033215ca9 | |||
| 1f4432c39b | |||
| a9f0fd9786 | |||
| 05bf05ca49 | |||
| 9bba28093c | |||
| 3519b718ea | |||
| cde173688b | |||
| 68668d7a5b | |||
| 55b914fe22 | |||
| b2c1b68d51 | |||
| 14270354d1 | |||
| 7b981cbf09 | |||
| 27596473c7 | |||
| 14421c6cde | |||
| 918b0b0d43 | |||
| 4cd20a464d | |||
| 1c628c6c1a | |||
| a55af6fa00 | |||
| 824390e33c | |||
| 2171e99416 | |||
| b6ad5a6ed3 | |||
| b49505be8e | |||
| 2ed6ec53d9 | |||
| 30b52d7d57 | |||
| d753daf758 | |||
| 74cfc5b906 | |||
| 58910d2ce6 | |||
| bd258fe970 | |||
| 68bf5665f0 | |||
| a57e625b17 | |||
| d7da15c826 | |||
| ac7c2155e4 | |||
| bccf5aed8a | |||
| 5bb73ff6cb | |||
| b772fe2191 | |||
| 28f072da31 | |||
| 34e143cd66 | |||
| e73db7e0d7 | |||
| f120341c9c | |||
| 54cd73eb85 | |||
| 20bd55c3a5 | |||
| 403d7fcf80 | |||
| f5761b1f2d | |||
| d4ade36184 | |||
| 982b43e4fd | |||
| b64ab18caa | |||
| 433594297a | |||
| dff89d2950 | |||
| 0bd8366254 | |||
| 2222897b3f | |||
| fc5f295de7 | |||
| 05c7712770 | |||
| 1737aff7c9 | |||
| 464222094f | |||
| 42d15623ca | |||
| 6c20263e2e | |||
| 50e637f9cd | |||
| 1605cddb40 | |||
| 75e496d461 | |||
| 9fcd5d473e | |||
| 5f1c814355 | |||
| 21d88f38d0 | |||
| fd27842a98 | |||
| bddb5eb935 | |||
| ffc85936f1 | |||
| 241c05ccac | |||
| 50e020b75d | |||
| d1c005ff6d | |||
| 96ee62c7a2 | |||
| c321ebfbcb | |||
| 1a44a954df | |||
| 4d8e334ef8 | |||
| b7be0f35f8 | |||
| 97c1f7a021 | |||
| f362a1618d | |||
| 43703d45d2 | |||
| f06412e628 | |||
| dc2bf4b442 | |||
| 8f1f19d4b9 | |||
| 064733d82b | |||
| da7169675c | |||
| 6de33fa13b | |||
| c1085921b5 | |||
| 18de5eea43 | |||
| 13214ba5d0 | |||
| 56f3a668f2 | |||
| 75f29f0dd9 | |||
| 345a702501 | |||
| d2baa40fb1 | |||
| 216361c640 | |||
| 904f4b8523 | |||
| 59337a9050 | |||
| 0db49991cb | |||
| 34bf87c410 | |||
| f37ec32981 | |||
| f8a4d2a7b5 | |||
| cefef61d0e | |||
| 3090718ca3 | |||
| 7eb7f49f52 | |||
| c255b79da4 | |||
| 94d0eca0ca | |||
| bfdee6b692 | |||
| 732839e4bd | |||
| 98a0801a32 | |||
| e1f86d7ca0 | |||
| 5c7ceec37f | |||
| 723a3f4225 | |||
| b2d56e39e5 | |||
| 0e447a3a19 | |||
| e937f8710c | |||
| b1fce69153 | |||
| b092c24374 | |||
| 1678461f5a | |||
| 35e3156018 | |||
| 611fb28c58 | |||
| 10abb20117 | |||
| 6e96cff435 | |||
| fcf17d3292 | |||
| 4235440ff4 | |||
| 111dddda18 | |||
| 4b25dd1913 | |||
| 2453a6bff7 | |||
| f596e59bd8 | |||
| cb5127735f | |||
| 993f16a089 | |||
| 90ffb6cc8a | |||
| 17526da856 | |||
| eb597924b2 | |||
| e5b1912bfd | |||
| f19a91abfc | |||
| 7ff55f0013 | |||
| 17a901e8da | |||
| 1a19622add | |||
| 7cfeaff95b | |||
| a355b57623 | |||
| 14f7fc5e53 | |||
| 7f4c2d18c6 | |||
| d1fdfe10ea | |||
| 4546ebba98 | |||
| dd600c0b0c | |||
| 0a89784116 | |||
| cb66555a9c | |||
| 1a5aaf3fea | |||
| cff2d79c42 | |||
| ea85d5efc8 | |||
| c3f78f8bce | |||
| d54e14f1d9 | |||
| b7fcb13c33 | |||
| e1032b6a31 | |||
| ca46d11123 | |||
| f9bd19179f | |||
| ae8fc773cb | |||
| 8662211492 | |||
| 892283dc89 | |||
| 76b8f073e4 | |||
| fe7eb4cbe1 | |||
| 8c4c53339c | |||
| ef66c56c2b | |||
| e512d08439 | |||
| c6304add4a | |||
| 90d9cdd0e9 | |||
| bf719e0993 | |||
| 1726caa414 | |||
| 25af7e6cce | |||
| cbbbb4e9a5 | |||
| 450317d8fe | |||
| d62407cfbb | |||
| 09325aed89 | |||
| e0ce86b180 | |||
| 2c206e9ce5 | |||
| 80e4a2aa79 | |||
| 3da6af95bd | |||
| 62beb3a7ae | |||
| a7ec1485a7 | |||
| 992956db8e | |||
| 6fe0d75b38 | |||
| c379b9221a | |||
| effce1df15 | |||
| d57477aefa | |||
| 972d8602c2 | |||
| b3e627fb80 |
+39
-10
@@ -7,24 +7,53 @@ For more information about configuration options, please refer to the documentat
|
||||
|
||||
"""
|
||||
|
||||
# ==========================================
|
||||
# Global configs:
|
||||
USE_AZURE=False
|
||||
USE_AZURE_TOKEN_PROVIDER=False
|
||||
MAX_RETRY=10
|
||||
RETRY_WAIT_SECONDS=20
|
||||
# ==========================================
|
||||
|
||||
# LLM API Setting:
|
||||
OPENAI_API_KEY=<your_api_key>
|
||||
CHAT_MODEL=gpt-4-turbo
|
||||
CHAT_MAX_TOKENS=3000
|
||||
CHAT_TEMPERATURE=0.7
|
||||
|
||||
# ==========================================
|
||||
# Backend Configuration
|
||||
# ==========================================
|
||||
# BACKEND=rdagent.oai.backend.LiteLLMAPIBackend
|
||||
# ==========================================
|
||||
|
||||
# ==========================================
|
||||
# Backend Configuration (choose one)
|
||||
# ==========================================
|
||||
|
||||
# 1. Set universal API key
|
||||
# CHAT_MODEL="gpt-4o"
|
||||
# EMBEDDING_MODEL="text-embedding-3-small"
|
||||
# OPENAI_API_BASE="https://your-endpoint.com/v1"
|
||||
# OPENAI_API_KEY="sk-your-api-key-here"
|
||||
|
||||
# 2. Set separate API KEY
|
||||
# Chat configuration
|
||||
OPENAI_API_KEY="sk-chat-key"
|
||||
OPENAI_API_BASE="https://xxx-litellm.com/v1"
|
||||
CHAT_MODEL='gpt-4o'
|
||||
|
||||
# Embedding configuration (using other service)
|
||||
# Use siliconflow as example, pay attention to the litellm_proxy prefix
|
||||
LITELLM_PROXY_API_KEY="sk-embedding-service-key"
|
||||
LITELLM_PROXY_API_BASE="https://api.siliconflow.cn/v1"
|
||||
EMBEDDING_MODEL="litellm_proxy/BAAI/bge-large-en-v1.5"
|
||||
# ==========================================
|
||||
|
||||
# ==========================================
|
||||
# Other Configuration
|
||||
# ==========================================
|
||||
# CHAT_AZURE_API_BASE=<for_Azure_user>
|
||||
# CHAT_AZURE_API_VERSION=<for_Azure_user>
|
||||
|
||||
EMBEDDING_MODEL=text-embedding-3-small
|
||||
# EMBEDDING_AZURE_API_BASE=<for_Azure_user>
|
||||
# EMBEDDING_AZURE_API_VERSION=<for_Azure_user>
|
||||
|
||||
# Cache Setting (Optional):
|
||||
|
||||
# Senario Configs:
|
||||
# USE_CHAT_CACHE=True
|
||||
# USE_EMBEDDING_CACHE=True
|
||||
# Senario Configs:
|
||||
# ==========================================
|
||||
@@ -20,14 +20,12 @@
|
||||
|
||||
## How Has This Been Tested?
|
||||
<!--- Put an `x` in all the boxes that apply: --->
|
||||
- [ ] Pass the test by running: `pytest qlib/tests/test_all_pipeline.py` under upper directory of `qlib`.
|
||||
- [ ] If you are adding a new feature, test on your own test scripts.
|
||||
|
||||
<!--- **ATTENTION**: If you are adding a new feature, please make sure your codes are **correctly tested**. If our test scripts do not cover your cases, please provide your own test scripts under the `tests` folder and test them. More information about test scripts can be found [here](https://docs.python.org/3/library/unittest.html#basic-example), or you could refer to those we provide under the `tests` folder. -->
|
||||
|
||||
## Screenshots of Test Results (if appropriate):
|
||||
1. Pipeline test:
|
||||
2. Your own tests:
|
||||
1. Your own tests:
|
||||
|
||||
## Types of changes
|
||||
<!--- What types of changes does your code introduce? Put an `x` in all the boxes that apply: -->
|
||||
|
||||
@@ -21,7 +21,7 @@ jobs:
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v3
|
||||
uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '16'
|
||||
|
||||
|
||||
+11
-3
@@ -4,6 +4,7 @@
|
||||
Pipfile
|
||||
public
|
||||
release-notes.md
|
||||
typescript*
|
||||
|
||||
# Byte-compiled / optimized / DLL files
|
||||
__pycache__/
|
||||
@@ -64,7 +65,7 @@ coverage.xml
|
||||
|
||||
# Django stuff:
|
||||
*.log
|
||||
/log/
|
||||
/log*/
|
||||
local_settings.py
|
||||
db.sqlite3
|
||||
db.sqlite3-journal
|
||||
@@ -110,7 +111,8 @@ celerybeat.pid
|
||||
*.sage.py
|
||||
|
||||
# Environments
|
||||
.env
|
||||
.env*
|
||||
*.env
|
||||
.venv
|
||||
^env/
|
||||
venv/
|
||||
@@ -169,4 +171,10 @@ mlruns/
|
||||
|
||||
# shell script
|
||||
*.out
|
||||
*.sh
|
||||
/*.sh
|
||||
.aider*
|
||||
rdagent/app/benchmark/factor/example.json
|
||||
|
||||
# UI Server resources
|
||||
videos/
|
||||
static/
|
||||
@@ -10,6 +10,18 @@ build:
|
||||
os: ubuntu-22.04
|
||||
tools:
|
||||
python: "3.10"
|
||||
# During the build process, you need to fetch tags, and since the default command to read the docs only pulls shallow code, it will cause an error.
|
||||
# So we added the `git fetch --tags --unshallow || true` command to fetch the full tag record.
|
||||
# Adding this command overrides the default command, so we copied it over to make sure the build was successful.
|
||||
commands:
|
||||
- python -mvirtualenv $READTHEDOCS_VIRTUALENV_PATH
|
||||
- python -m pip install --upgrade --no-cache-dir pip setuptools
|
||||
- python -m pip install --upgrade --no-cache-dir sphinx
|
||||
- python -m pip install --exists-action=w --no-cache-dir -r requirements/docs.txt
|
||||
- python -m pip install --upgrade --upgrade-strategy only-if-needed --no-cache-dir .
|
||||
- git fetch --tags --unshallow || true
|
||||
- mkdir -p $READTHEDOCS_OUTPUT/html/
|
||||
- python -m sphinx -T -b html -d _build/doctrees -D language=en ./docs $READTHEDOCS_OUTPUT/html
|
||||
|
||||
# Build documentation in the docs/ directory with Sphinx
|
||||
sphinx:
|
||||
|
||||
+235
@@ -1,5 +1,240 @@
|
||||
# Changelog
|
||||
|
||||
## [0.5.0](https://github.com/microsoft/RD-Agent/compare/v0.4.0...v0.5.0) (2025-06-18)
|
||||
|
||||
|
||||
### Features
|
||||
|
||||
* add a check for whether values in score_df are NaN ([#756](https://github.com/microsoft/RD-Agent/issues/756)) ([d9cc780](https://github.com/microsoft/RD-Agent/commit/d9cc78098beb27f3a1bf2f2d461302db177b7d41))
|
||||
* add competition level filter and extract constants to utils ([#869](https://github.com/microsoft/RD-Agent/issues/869)) ([b40b605](https://github.com/microsoft/RD-Agent/commit/b40b6055368e6c72d8435352104b1c281b06da7f))
|
||||
* add DocDev for auto-generating workspace documentation ([#781](https://github.com/microsoft/RD-Agent/issues/781)) ([bcba6ea](https://github.com/microsoft/RD-Agent/commit/bcba6eac32684ebb267c93b4e85dbfa9561d15d1))
|
||||
* add drafting pipeline ([#832](https://github.com/microsoft/RD-Agent/issues/832)) ([efedddf](https://github.com/microsoft/RD-Agent/commit/efedddf39bc19221fdffc2e39ee0a09097fc82b0))
|
||||
* add last_exp_fb to DSTrace and update feedback retrieval usage ([#910](https://github.com/microsoft/RD-Agent/issues/910)) ([10531fd](https://github.com/microsoft/RD-Agent/commit/10531fda9438c6915b26d5013bd2413e1333ceb9))
|
||||
* add mlflow logger in RD loop to log ([#815](https://github.com/microsoft/RD-Agent/issues/815)) ([b91b54f](https://github.com/microsoft/RD-Agent/commit/b91b54f355c26b751087d0c14774f466e82866de))
|
||||
* add naive experiment generator and update proposal configurations ([#759](https://github.com/microsoft/RD-Agent/issues/759)) ([75494f4](https://github.com/microsoft/RD-Agent/commit/75494f4fed5bc845acfd7f7bacef385f0f96c514))
|
||||
* add RD-Agent-Quant scenario ([#838](https://github.com/microsoft/RD-Agent/issues/838)) ([6e42d52](https://github.com/microsoft/RD-Agent/commit/6e42d523a85df67aa13927abbf0894564c71880e))
|
||||
* add reasoning_effort parameter to LiteLLMAPIBackend and LLMSett… ([#754](https://github.com/microsoft/RD-Agent/issues/754)) ([113889f](https://github.com/microsoft/RD-Agent/commit/113889fefe9b09aaea1b564704c81664b8f77ec5))
|
||||
* add reviewer in feedback ([#765](https://github.com/microsoft/RD-Agent/issues/765)) ([1a95bee](https://github.com/microsoft/RD-Agent/commit/1a95bee6aa6bc6f45fdeb484f3a6f81caa273038))
|
||||
* advanced checkpoint selectors ([#790](https://github.com/microsoft/RD-Agent/issues/790)) ([50ea033](https://github.com/microsoft/RD-Agent/commit/50ea0336e93d8cb39fb871e81a3f61abdf293bc7))
|
||||
* archive python and csv files in workspace to maintain results ([#814](https://github.com/microsoft/RD-Agent/issues/814)) ([67d0e01](https://github.com/microsoft/RD-Agent/commit/67d0e01e7c9237da1371d93cbf9d86f5f46faac4))
|
||||
* checkpoint selection ([#744](https://github.com/microsoft/RD-Agent/issues/744)) ([a15a06a](https://github.com/microsoft/RD-Agent/commit/a15a06ad643977db59d7cac9da52e637cf80395a))
|
||||
* custom data ([#810](https://github.com/microsoft/RD-Agent/issues/810)) ([6322916](https://github.com/microsoft/RD-Agent/commit/632291608cf605bd8bcfcab0017824823bdecdb8))
|
||||
* dump model ([#776](https://github.com/microsoft/RD-Agent/issues/776)) ([b49481e](https://github.com/microsoft/RD-Agent/commit/b49481e073e6f536d2b1b3bd2d01229ed05abdea))
|
||||
* enable to set different version of idea-proposal for multi traces ([#895](https://github.com/microsoft/RD-Agent/issues/895)) ([236c28f](https://github.com/microsoft/RD-Agent/commit/236c28f29c6bc5da62129632e464bbc32056ebdb))
|
||||
* enhance compatibility with more LLM models ([#905](https://github.com/microsoft/RD-Agent/issues/905)) ([8800624](https://github.com/microsoft/RD-Agent/commit/8800624ad4749d6e798785a082c9f94c306792ef))
|
||||
* idea pool integrated to exp_gen & add timer to RD-Agent & pause-resume to RD-loops ([#795](https://github.com/microsoft/RD-Agent/issues/795)) ([e62aefa](https://github.com/microsoft/RD-Agent/commit/e62aefa56e34ff45a8ed033f7bf28b95c8e63656))
|
||||
* joblib cache ([#749](https://github.com/microsoft/RD-Agent/issues/749)) ([83a0411](https://github.com/microsoft/RD-Agent/commit/83a041148ff908871b1906f9e6889d80ab513412))
|
||||
* log api status to mlflow ([#860](https://github.com/microsoft/RD-Agent/issues/860)) ([049921b](https://github.com/microsoft/RD-Agent/commit/049921beb0b4ed0ba1ab7508d9857d2c1e729349))
|
||||
* log reaching max time limit before breaking CoSTEER evolution ([#921](https://github.com/microsoft/RD-Agent/issues/921)) ([837fff2](https://github.com/microsoft/RD-Agent/commit/837fff29096fefe1369d386ef8a860395b737173))
|
||||
* merge failed and successful traces together ([#766](https://github.com/microsoft/RD-Agent/issues/766)) ([3a2aa8c](https://github.com/microsoft/RD-Agent/commit/3a2aa8cf0102647950b2dfc0007c118b0c799cd4))
|
||||
* merge selectively ([#888](https://github.com/microsoft/RD-Agent/issues/888)) ([06ba314](https://github.com/microsoft/RD-Agent/commit/06ba314ff0f91e7e78e8d456c719ac3194a8c774))
|
||||
* multi-trace online merge ([#886](https://github.com/microsoft/RD-Agent/issues/886)) ([2112d67](https://github.com/microsoft/RD-Agent/commit/2112d676d0938de6fea163b2e5eb9c36771e7041))
|
||||
* new proposal (structured outputs) prompts ([#887](https://github.com/microsoft/RD-Agent/issues/887)) ([150796a](https://github.com/microsoft/RD-Agent/commit/150796aaa72eaa5037fd7db8e785058fbc4d4967))
|
||||
* parallel loop running based on asyncio ([#932](https://github.com/microsoft/RD-Agent/issues/932)) ([c63e207](https://github.com/microsoft/RD-Agent/commit/c63e2071f3179feef69f88061c0172cb5c3157f2))
|
||||
* propose hypothesis across multiple parts in pipeline ([#827](https://github.com/microsoft/RD-Agent/issues/827)) ([acb0e21](https://github.com/microsoft/RD-Agent/commit/acb0e21a331410d044849e12e2887f41e5ff1c3a))
|
||||
* pull image with progress ([#777](https://github.com/microsoft/RD-Agent/issues/777)) ([5cad086](https://github.com/microsoft/RD-Agent/commit/5cad0860204ede974533dc7bdc9808cfd135fa24))
|
||||
* raise error when timeout in api call ([#793](https://github.com/microsoft/RD-Agent/issues/793)) ([eafd4df](https://github.com/microsoft/RD-Agent/commit/eafd4dfc6263f19a8cdaf27498a1d07b43815306))
|
||||
* raise policy violation ([#894](https://github.com/microsoft/RD-Agent/issues/894)) ([5b9d007](https://github.com/microsoft/RD-Agent/commit/5b9d0072aebe15369e9a0010af83e71684baeae7))
|
||||
* reanalyze competition info & pipeline coding evaluator prompt ([#837](https://github.com/microsoft/RD-Agent/issues/837)) ([f7b5258](https://github.com/microsoft/RD-Agent/commit/f7b52580080c75d311355bcc6193b49495801809))
|
||||
* refine merge ([#842](https://github.com/microsoft/RD-Agent/issues/842)) ([99463b4](https://github.com/microsoft/RD-Agent/commit/99463b46819b3a0dcb2bb12a823a9cdf7ec560b4))
|
||||
* refine prompt ([#760](https://github.com/microsoft/RD-Agent/issues/760)) ([a91b182](https://github.com/microsoft/RD-Agent/commit/a91b182c4c9510eb34e4aab956588e909fa5d70b))
|
||||
* replace hard-coded cache paths with dynamic cache_path config ([#952](https://github.com/microsoft/RD-Agent/issues/952)) ([db56894](https://github.com/microsoft/RD-Agent/commit/db568947f1084a80d603718f5a13fdbd72b90a47))
|
||||
* revert draft stage into a soft decay in hypothesis selection ([#849](https://github.com/microsoft/RD-Agent/issues/849)) ([d41db0c](https://github.com/microsoft/RD-Agent/commit/d41db0ca357b07091825ebd9d18c303b6db3cc6a))
|
||||
* trace merging ([#836](https://github.com/microsoft/RD-Agent/issues/836)) ([a3d5473](https://github.com/microsoft/RD-Agent/commit/a3d547369e408a05cff570c1239b6320be40418d))
|
||||
* truncate by time ([#863](https://github.com/microsoft/RD-Agent/issues/863)) ([2b9427a](https://github.com/microsoft/RD-Agent/commit/2b9427ae036ffe1e28a717502f45500fe91fe5ac))
|
||||
* update prompt to improve json respond format of some LLM models ([#928](https://github.com/microsoft/RD-Agent/issues/928)) ([0b84709](https://github.com/microsoft/RD-Agent/commit/0b84709e59c7abb9754961cd17cc9673fcf508aa))
|
||||
* using different chat model in different part ([#822](https://github.com/microsoft/RD-Agent/issues/822)) ([c052ea6](https://github.com/microsoft/RD-Agent/commit/c052ea6d1f8948183a4a6ebc873ec01b57373cce))
|
||||
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* 'DSProposalV2ExpGen' object has no attribute 'COMPONENT_TASK_MAP… ([#950](https://github.com/microsoft/RD-Agent/issues/950)) ([e353895](https://github.com/microsoft/RD-Agent/commit/e353895251f231fee85abdcb1b22b022a577af77))
|
||||
* adapting UI to mock trace ([#841](https://github.com/microsoft/RD-Agent/issues/841)) ([8a5754c](https://github.com/microsoft/RD-Agent/commit/8a5754c9b9c9410d0943aeed777a93c13422e54a))
|
||||
* add missing semicolon after chmod in env shell command ([#955](https://github.com/microsoft/RD-Agent/issues/955)) ([1128eaa](https://github.com/microsoft/RD-Agent/commit/1128eaa89ec1dcab4a05ef50d64c7f7e6aae88a8))
|
||||
* add time to timer when api timeout bug ([#826](https://github.com/microsoft/RD-Agent/issues/826)) ([f45d6ae](https://github.com/microsoft/RD-Agent/commit/f45d6ae6595c1c39b389485b637a0ae53ffc8782))
|
||||
* add wait_retry to exp_gen v2 ([#783](https://github.com/microsoft/RD-Agent/issues/783)) ([b9fb7cf](https://github.com/microsoft/RD-Agent/commit/b9fb7cf4e3070062d91b5b67d0f10d6266b45142))
|
||||
* adjust ds_trace lookup and add stderr redirect to mlebench command ([#853](https://github.com/microsoft/RD-Agent/issues/853)) ([4e53108](https://github.com/microsoft/RD-Agent/commit/4e53108e020db719b39cba3a67e0c6dae3de19cf))
|
||||
* align competion_full_desc and scenario_all_desc, remove redundant info in problems proposal ([#808](https://github.com/microsoft/RD-Agent/issues/808)) ([76d8536](https://github.com/microsoft/RD-Agent/commit/76d8536d9ec53952383019306781d49cb3e9f75c))
|
||||
* bug fix in timer start ([#807](https://github.com/microsoft/RD-Agent/issues/807)) ([9af7161](https://github.com/microsoft/RD-Agent/commit/9af7161eb57bdd2e24b072335e9d185951c32472))
|
||||
* bug in problem identification ([#806](https://github.com/microsoft/RD-Agent/issues/806)) ([e1d5a29](https://github.com/microsoft/RD-Agent/commit/e1d5a2914046476f2f10d5884ed3c3ff956d65ff))
|
||||
* conda error information ([#941](https://github.com/microsoft/RD-Agent/issues/941)) ([fd39a94](https://github.com/microsoft/RD-Agent/commit/fd39a947763fb4a9be87b907c399bebe384df505))
|
||||
* default cost to NaN when calculation fails in LiteLLM backend ([#912](https://github.com/microsoft/RD-Agent/issues/912)) ([51a4048](https://github.com/microsoft/RD-Agent/commit/51a4048129cbfbc3b84bcf50fd8866fafb3e2da3))
|
||||
* ds trace ([#929](https://github.com/microsoft/RD-Agent/issues/929)) ([127e441](https://github.com/microsoft/RD-Agent/commit/127e441602e21a46d6313ff39133ab8ca841937e))
|
||||
* duplicate model names test in pipeline coder & runner ([#763](https://github.com/microsoft/RD-Agent/issues/763)) ([be3ee9d](https://github.com/microsoft/RD-Agent/commit/be3ee9da9882edda3c06ff7d1099d1bbda2203c3))
|
||||
* filter system metadata dirs and init missing DSTrace attribute ([#946](https://github.com/microsoft/RD-Agent/issues/946)) ([10050ef](https://github.com/microsoft/RD-Agent/commit/10050ef368ae7ec07cbf20ac4e52e21c2875eaab))
|
||||
* fix a bug in docker result extraction ([#824](https://github.com/microsoft/RD-Agent/issues/824)) ([e1c0f98](https://github.com/microsoft/RD-Agent/commit/e1c0f9826abcbc11dda215a600a2637c9ac6e984))
|
||||
* fix competition metric direction ([#784](https://github.com/microsoft/RD-Agent/issues/784)) ([3be0057](https://github.com/microsoft/RD-Agent/commit/3be0057556f46c899065ee1c7f9bafe33e79249c))
|
||||
* fix model input shape bug and costeer_model bug ([#821](https://github.com/microsoft/RD-Agent/issues/821)) ([b34bd89](https://github.com/microsoft/RD-Agent/commit/b34bd895d6d9c326aab85856a15be0cb72b2c4c8))
|
||||
* fix some minor bugs ([#758](https://github.com/microsoft/RD-Agent/issues/758)) ([963f96e](https://github.com/microsoft/RD-Agent/commit/963f96e5596bee04074135c2a0e31a8adc39ad8c))
|
||||
* fix some minor bugs in qlib scenario ([#817](https://github.com/microsoft/RD-Agent/issues/817)) ([79962a7](https://github.com/microsoft/RD-Agent/commit/79962a7ca40c77a3997a68da9ad1b5ab16728483))
|
||||
* fix the bug in the regular expression matching for stdout ([#890](https://github.com/microsoft/RD-Agent/issues/890)) ([ee57e37](https://github.com/microsoft/RD-Agent/commit/ee57e37a22af874b262c033d1606dbe7799706db))
|
||||
* fix the bug of Exceed-LLM-Context in online merge of multi-tarce ([#892](https://github.com/microsoft/RD-Agent/issues/892)) ([f760a3e](https://github.com/microsoft/RD-Agent/commit/f760a3eff7bd927a31e4958ed2f706312e83e3e3))
|
||||
* fix the problems weights bug ([#898](https://github.com/microsoft/RD-Agent/issues/898)) ([013d79f](https://github.com/microsoft/RD-Agent/commit/013d79f12060e908aeb57c3eb1bb56eea86df086))
|
||||
* fixed CI execution failures caused by document builds ([#857](https://github.com/microsoft/RD-Agent/issues/857)) ([5c116b2](https://github.com/microsoft/RD-Agent/commit/5c116b24ce727f6ed9ef39d5aa5b60442038c344))
|
||||
* get_metric_direction for aerial-cactus-identification ([#970](https://github.com/microsoft/RD-Agent/issues/970)) ([70dc62d](https://github.com/microsoft/RD-Agent/commit/70dc62de5fbd4272ecda1b6fcbcf898b3624a991))
|
||||
* import path of T ([#787](https://github.com/microsoft/RD-Agent/issues/787)) ([ac008a6](https://github.com/microsoft/RD-Agent/commit/ac008a61d03b4737ab3d994024e922839d8f3fe1))
|
||||
* improve eval alignment check (e.g. small-scale finetuning) ([#802](https://github.com/microsoft/RD-Agent/issues/802)) ([d391578](https://github.com/microsoft/RD-Agent/commit/d3915788082de640a4ce1eea6d2e607319b89c3e))
|
||||
* improve file tree and _walk symlink handling ([#877](https://github.com/microsoft/RD-Agent/issues/877)) ([516cb69](https://github.com/microsoft/RD-Agent/commit/516cb69357483ddd99f84b221a056d8491c34f9b))
|
||||
* log info ([#965](https://github.com/microsoft/RD-Agent/issues/965)) ([f1dbc21](https://github.com/microsoft/RD-Agent/commit/f1dbc2100498e22c8e5edbb2e4563c99c3d54775))
|
||||
* main bug ([#938](https://github.com/microsoft/RD-Agent/issues/938)) ([c6d34d6](https://github.com/microsoft/RD-Agent/commit/c6d34d67b8aedf5496bf6a875915ce657fc58448))
|
||||
* non-exist variable test_eval.py ([#847](https://github.com/microsoft/RD-Agent/issues/847)) ([4948c38](https://github.com/microsoft/RD-Agent/commit/4948c38560f4cf021d9354b201b22dfa5ccb9441))
|
||||
* refine feedback prompt ([#901](https://github.com/microsoft/RD-Agent/issues/901)) ([12bb2c4](https://github.com/microsoft/RD-Agent/commit/12bb2c4a1494b9aa29962905abb5e433a60eb716))
|
||||
* refine the time/memory constraints prompt in hypothesis proposal ([#856](https://github.com/microsoft/RD-Agent/issues/856)) ([51ce8ef](https://github.com/microsoft/RD-Agent/commit/51ce8ef84b4fe6590ce20599a56eee596f2f04e6))
|
||||
* Set PYTHONPATH in env.run_ret_code call in FBWorkspace class ([#755](https://github.com/microsoft/RD-Agent/issues/755)) ([68b5018](https://github.com/microsoft/RD-Agent/commit/68b501889caca754f27b57d9ab6f72184e93b15c))
|
||||
* task_gen for better understanding ([#752](https://github.com/microsoft/RD-Agent/issues/752)) ([6bfc1e5](https://github.com/microsoft/RD-Agent/commit/6bfc1e570449ee69ac110a4ced9a7cecbc0e6a73))
|
||||
* trace list but ([#852](https://github.com/microsoft/RD-Agent/issues/852)) ([32cdc57](https://github.com/microsoft/RD-Agent/commit/32cdc575bde103d71a358d4d99bd413076328ebd))
|
||||
* typo in workflow ([#861](https://github.com/microsoft/RD-Agent/issues/861)) ([0e54c9f](https://github.com/microsoft/RD-Agent/commit/0e54c9fe41d25a4cc45ab9e61bb2c2c01b854751))
|
||||
* update DS env setup with competition volume and timeout ([#878](https://github.com/microsoft/RD-Agent/issues/878)) ([816ada0](https://github.com/microsoft/RD-Agent/commit/816ada096afabe90578672b0e61b656802a30b62))
|
||||
* update feedback.py ([#772](https://github.com/microsoft/RD-Agent/issues/772)) ([133778c](https://github.com/microsoft/RD-Agent/commit/133778c67ee3349f1c2fe029bcf6a9ee14568efe))
|
||||
* update metric direction to return bool ([#791](https://github.com/microsoft/RD-Agent/issues/791)) ([0bf365e](https://github.com/microsoft/RD-Agent/commit/0bf365e7830aa86d2350b9d1c47410af46b3a7e8))
|
||||
* update runner max loop to 1 in DS scenario ([#820](https://github.com/microsoft/RD-Agent/issues/820)) ([3da378e](https://github.com/microsoft/RD-Agent/commit/3da378e986e8b776a17dbc694d29ef211192ed3e))
|
||||
* use fallback messages for missing submission and scores files ([#882](https://github.com/microsoft/RD-Agent/issues/882)) ([898fdea](https://github.com/microsoft/RD-Agent/commit/898fdeae80801d537ebc5c4a3b7df9de74c3403a))
|
||||
* use simple stdout and stderr ([#966](https://github.com/microsoft/RD-Agent/issues/966)) ([0b1c445](https://github.com/microsoft/RD-Agent/commit/0b1c445f1f0c212887ffff9f8fac44236df3607c))
|
||||
* use trace count as index ([#909](https://github.com/microsoft/RD-Agent/issues/909)) ([b87de56](https://github.com/microsoft/RD-Agent/commit/b87de56e54b206b3aada53850804474eff80b96d))
|
||||
* wrong variable test_eval.py ([#846](https://github.com/microsoft/RD-Agent/issues/846)) ([808ea6c](https://github.com/microsoft/RD-Agent/commit/808ea6cba541e60c35dd283cee9098ce46f2a59e))
|
||||
|
||||
## [0.4.0](https://github.com/microsoft/RD-Agent/compare/v0.3.0...v0.4.0) (2025-04-04)
|
||||
|
||||
|
||||
### Features
|
||||
|
||||
* (Kaggle) add base template for competition: tabular-playground-series-may-2022 ([#481](https://github.com/microsoft/RD-Agent/issues/481)) ([f3405ca](https://github.com/microsoft/RD-Agent/commit/f3405ca732eb0ddca8e18ea72f69cbd86055c4ab))
|
||||
* a unified CoSTEER to fit more scenarios ([#491](https://github.com/microsoft/RD-Agent/issues/491)) ([cddbd02](https://github.com/microsoft/RD-Agent/commit/cddbd02e3ad3ccf6ad01443777319dc5c7eb08a7))
|
||||
* add a new competition ([#474](https://github.com/microsoft/RD-Agent/issues/474)) ([2fc0d77](https://github.com/microsoft/RD-Agent/commit/2fc0d77c485a31f647e21f4578e2e326f7032964))
|
||||
* add a tool to enable saving workspace files into a specific folder ([#728](https://github.com/microsoft/RD-Agent/issues/728)) ([bca864b](https://github.com/microsoft/RD-Agent/commit/bca864b7edeafe3f88405efb695ca8acad6252f8))
|
||||
* add baseline score stat ([#590](https://github.com/microsoft/RD-Agent/issues/590)) ([2948026](https://github.com/microsoft/RD-Agent/commit/2948026c390d067b643f8c8247c1447f1dc023e4))
|
||||
* add configurable volume mode for Docker volumes in env.py ([#537](https://github.com/microsoft/RD-Agent/issues/537)) ([642a022](https://github.com/microsoft/RD-Agent/commit/642a02239431411b91959f23e69b454997ca75d5))
|
||||
* add constraint labels for semantic search ([#680](https://github.com/microsoft/RD-Agent/issues/680)) ([0584cfc](https://github.com/microsoft/RD-Agent/commit/0584cfcd13ca1a62c85390ea2ee7574370748d31))
|
||||
* add cross validation to workflow ([#700](https://github.com/microsoft/RD-Agent/issues/700)) ([82e9b00](https://github.com/microsoft/RD-Agent/commit/82e9b00be62b01673353a7aaa3ab0e2e3ecaf3ca))
|
||||
* add describe_data_folder_v2 ([#738](https://github.com/microsoft/RD-Agent/issues/738)) ([bc8e846](https://github.com/microsoft/RD-Agent/commit/bc8e8460e0246321792ff3347b1b8905416ad075))
|
||||
* add do_truncate control for the load function ([#656](https://github.com/microsoft/RD-Agent/issues/656)) ([2b960a5](https://github.com/microsoft/RD-Agent/commit/2b960a58dfdeba69522a0f72ecf0975bb6ae87ee))
|
||||
* add do_truncate control for the load function ([#656](https://github.com/microsoft/RD-Agent/issues/656)) ([2b960a5](https://github.com/microsoft/RD-Agent/commit/2b960a58dfdeba69522a0f72ecf0975bb6ae87ee))
|
||||
* add eda to data science scenario ([#639](https://github.com/microsoft/RD-Agent/issues/639)) ([35aa479](https://github.com/microsoft/RD-Agent/commit/35aa479f00edf118d43ec228e0a84c155332957a))
|
||||
* add hypothesis guidelines and rule-based ranking ([#746](https://github.com/microsoft/RD-Agent/issues/746)) ([c077b82](https://github.com/microsoft/RD-Agent/commit/c077b8239cc72904c4bc450845ed2a11aa5445f0))
|
||||
* Add line length limit to shrink_text function and settings ([#715](https://github.com/microsoft/RD-Agent/issues/715)) ([75ed5e1](https://github.com/microsoft/RD-Agent/commit/75ed5e1c2ce1bf20bb55190c10a4134e04694d2b))
|
||||
* add loop_n parameter to the main loop ([#611](https://github.com/microsoft/RD-Agent/issues/611)) ([778c166](https://github.com/microsoft/RD-Agent/commit/778c166962250e3b9e7ad85de37f62297d370b45))
|
||||
* add max time config to costeer in data science ([#645](https://github.com/microsoft/RD-Agent/issues/645)) ([534686c](https://github.com/microsoft/RD-Agent/commit/534686c2ba7d9fa979c0762ad3177c36f6d7f4cb))
|
||||
* add mlebench submission validitor ([#545](https://github.com/microsoft/RD-Agent/issues/545)) ([712d94a](https://github.com/microsoft/RD-Agent/commit/712d94a7d6f22187fc3d18bd434e71ec6997aa9f))
|
||||
* add model removal and adjust some framework logic ([#681](https://github.com/microsoft/RD-Agent/issues/681)) ([1edf881](https://github.com/microsoft/RD-Agent/commit/1edf881c63512d351c0dd074d7a1c0965ff3119b))
|
||||
* add output_path to load function of LoopBase ([#628](https://github.com/microsoft/RD-Agent/issues/628)) ([dd33726](https://github.com/microsoft/RD-Agent/commit/dd33726ac5de75dc2030d193d457d59490b3361e))
|
||||
* add pipeline coder ([#742](https://github.com/microsoft/RD-Agent/issues/742)) ([759f295](https://github.com/microsoft/RD-Agent/commit/759f295dbf1224e177006e72d694e42dd6f372b6))
|
||||
* add rank into report (mle_summary) ([#665](https://github.com/microsoft/RD-Agent/issues/665)) ([13f7922](https://github.com/microsoft/RD-Agent/commit/13f7922aaae9e4143aac4ad08ec1c556c2faf04e))
|
||||
* add restart and fix unzip ([#538](https://github.com/microsoft/RD-Agent/issues/538)) ([ed2c7d1](https://github.com/microsoft/RD-Agent/commit/ed2c7d175f1f44ca06ad7a63b08da12f6c4df9ab))
|
||||
* add retry mechanism with wait_retry decorator and refactor diff generation ([#572](https://github.com/microsoft/RD-Agent/issues/572)) ([de1cd72](https://github.com/microsoft/RD-Agent/commit/de1cd72f068ebd1e1bd5bc2ad2b12ae484d54831))
|
||||
* add the shape of the CSV to the dataset description ([#561](https://github.com/microsoft/RD-Agent/issues/561)) ([a10c881](https://github.com/microsoft/RD-Agent/commit/a10c881bd86796e6167257ad26dd165f7e46d813))
|
||||
* add timeout settings and cleanup step in data science runner ([#539](https://github.com/microsoft/RD-Agent/issues/539)) ([295abd5](https://github.com/microsoft/RD-Agent/commit/295abd56f7b58055bd27b247dfed47eb85e9b0cd))
|
||||
* add type checker to api backend & align litellm and old backend ([#647](https://github.com/microsoft/RD-Agent/issues/647)) ([d38eae9](https://github.com/microsoft/RD-Agent/commit/d38eae986a0ba69d71288fa09fcc21e227551a02))
|
||||
* align mlebench data and evaluation & several fix on kaggle workflow ([#477](https://github.com/microsoft/RD-Agent/issues/477)) ([f6c522b](https://github.com/microsoft/RD-Agent/commit/f6c522b651db3c1f6af6815347589917f46e433a))
|
||||
* **backend:** integrate LiteLLM API Backend ([#564](https://github.com/microsoft/RD-Agent/issues/564)) ([f477687](https://github.com/microsoft/RD-Agent/commit/f4776879c76a213d53875b307c94be1ea5cfd9ba))
|
||||
* base data science scenario UI ([#525](https://github.com/microsoft/RD-Agent/issues/525)) ([39917b3](https://github.com/microsoft/RD-Agent/commit/39917b354b22a8488a17396fe2245cb41e3def03))
|
||||
* condaenv & full docker env ([#668](https://github.com/microsoft/RD-Agent/issues/668)) ([084dd6d](https://github.com/microsoft/RD-Agent/commit/084dd6d748a89492ea0888acb316b9bb9efeb62f))
|
||||
* diff mode fix ([#569](https://github.com/microsoft/RD-Agent/issues/569)) ([0c509f5](https://github.com/microsoft/RD-Agent/commit/0c509f599ce19303b44d8192ec3eb634c24992d6))
|
||||
* display LLM prompt ([#676](https://github.com/microsoft/RD-Agent/issues/676)) ([8c93bba](https://github.com/microsoft/RD-Agent/commit/8c93bba82e185edcf4204cc574df5f41bcdfa9d2))
|
||||
* Dynamically find and use sample submission file in eval tests ([#542](https://github.com/microsoft/RD-Agent/issues/542)) ([5f12b44](https://github.com/microsoft/RD-Agent/commit/5f12b44c89dd26b250e914192f9beb2da38fb3ab))
|
||||
* end-to-end optimization ([#473](https://github.com/microsoft/RD-Agent/issues/473)) ([d41343a](https://github.com/microsoft/RD-Agent/commit/d41343a63d87bf3479f5ec30745ea788580495bf))
|
||||
* Enhance eval script with file cleanup and detailed submission checks ([#529](https://github.com/microsoft/RD-Agent/issues/529)) ([cf2ff92](https://github.com/microsoft/RD-Agent/commit/cf2ff9213d3a8b0fad64df7cae0c35f996d72e27))
|
||||
* exclude invalid session log folder ([#554](https://github.com/microsoft/RD-Agent/issues/554)) ([fa86e4d](https://github.com/microsoft/RD-Agent/commit/fa86e4d1805000e0e5779c662ccbb5273fda623c))
|
||||
* improve the framework's ability to adaptively adjust the model ([#629](https://github.com/microsoft/RD-Agent/issues/629)) ([93806f3](https://github.com/microsoft/RD-Agent/commit/93806f33a1e0f29a125e29303d4b984a9817c3c0))
|
||||
* independent use_azure_token_provider on chat and embedding ([#452](https://github.com/microsoft/RD-Agent/issues/452)) ([d223004](https://github.com/microsoft/RD-Agent/commit/d223004917692e231b251330cbc8676081d5a10d))
|
||||
* integrate azure deepseek r1 ([#591](https://github.com/microsoft/RD-Agent/issues/591)) ([e79ce5c](https://github.com/microsoft/RD-Agent/commit/e79ce5c38539138abe04eb9809fbde437e97bbb7))
|
||||
* kaggle refactor ([#489](https://github.com/microsoft/RD-Agent/issues/489)) ([1b057d0](https://github.com/microsoft/RD-Agent/commit/1b057d0d63a861fba4b3cb59c6c5fc1a0e3da383))
|
||||
* **kaggle:** several update in kaggle scenarios ([#476](https://github.com/microsoft/RD-Agent/issues/476)) ([245d211](https://github.com/microsoft/RD-Agent/commit/245d211dcbfb18ebcc554247a0e3a8dbecf6f3bd))
|
||||
* loader prompt & simplify YAML loading and update data loader specifications ([#736](https://github.com/microsoft/RD-Agent/issues/736)) ([86f8bbf](https://github.com/microsoft/RD-Agent/commit/86f8bbf15895e7c198f9bc395d055ca5f02a5bb6))
|
||||
* make spec optional ([#719](https://github.com/microsoft/RD-Agent/issues/719)) ([a16b70f](https://github.com/microsoft/RD-Agent/commit/a16b70ff34c66d7e1c4c7ff5236eca8e7d8abea9))
|
||||
* Make system prompt role customizable in LLM settings ([#632](https://github.com/microsoft/RD-Agent/issues/632)) ([e4acd92](https://github.com/microsoft/RD-Agent/commit/e4acd92cc5eec6db5c29cb2d4788020fb89099b7))
|
||||
* multi log folder, replace "epxx" in workspace path ([#555](https://github.com/microsoft/RD-Agent/issues/555)) ([8a69c9c](https://github.com/microsoft/RD-Agent/commit/8a69c9c9630860c9b644356e1f71654aea222328))
|
||||
* new exp gen v2 implementation ([#725](https://github.com/microsoft/RD-Agent/issues/725)) ([5dcc2d5](https://github.com/microsoft/RD-Agent/commit/5dcc2d5fa63bbe9ae8c4817d9b40b77600440edb))
|
||||
* new-york-city-taxi-fare-prediction_template ([#488](https://github.com/microsoft/RD-Agent/issues/488)) ([a9caab7](https://github.com/microsoft/RD-Agent/commit/a9caab7bc5dc86f395a008e523355922137aef17))
|
||||
* out spec change for o1-preview ([#666](https://github.com/microsoft/RD-Agent/issues/666)) ([22894bd](https://github.com/microsoft/RD-Agent/commit/22894bdbee26b9cad73646d2975857787e515f75))
|
||||
* refactor for general data science ([#498](https://github.com/microsoft/RD-Agent/issues/498)) ([7002dc4](https://github.com/microsoft/RD-Agent/commit/7002dc4981a4f72096b438d2fe4fd9ff268c54f3))
|
||||
* refine logic for qlib_factor_from_report ([#463](https://github.com/microsoft/RD-Agent/issues/463)) ([21348d8](https://github.com/microsoft/RD-Agent/commit/21348d89e0e0eec1b4fab4e7a497f1eb34b8fe72))
|
||||
* run benchmark on gpt-4o & llama 3.1 ([#497](https://github.com/microsoft/RD-Agent/issues/497)) ([64af0b5](https://github.com/microsoft/RD-Agent/commit/64af0b5529b687cce8b5b7a1893946e15edca626))
|
||||
* summary and UI update ([#581](https://github.com/microsoft/RD-Agent/issues/581)) ([efa51f9](https://github.com/microsoft/RD-Agent/commit/efa51f9c259a06fe219f3137f0a1005e50d2bfdd))
|
||||
* template changes for some kaggle competitions ([#484](https://github.com/microsoft/RD-Agent/issues/484)) ([2e38000](https://github.com/microsoft/RD-Agent/commit/2e38000091030811fc081d72016c7bbadf7efd50))
|
||||
* track and log accumulated completion cost in LiteLLMAPIBackend ([#727](https://github.com/microsoft/RD-Agent/issues/727)) ([b294a95](https://github.com/microsoft/RD-Agent/commit/b294a95e0b7b2ef96af355cebac92d9c87f3acab))
|
||||
* update prompts and descriptions for data science components ([#731](https://github.com/microsoft/RD-Agent/issues/731)) ([c20e226](https://github.com/microsoft/RD-Agent/commit/c20e226c3e7771c9fcd1c879a8937e4694dc03eb))
|
||||
* variable printing tool of data_science coder testing ([#658](https://github.com/microsoft/RD-Agent/issues/658)) ([116c061](https://github.com/microsoft/RD-Agent/commit/116c06190b01f0b621c021726a1be23458ab1154))
|
||||
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* a default conf in scen qlib ([#503](https://github.com/microsoft/RD-Agent/issues/503)) ([d64a228](https://github.com/microsoft/RD-Agent/commit/d64a228525cbedd7687c1e06132eacd0d0647697))
|
||||
* a small bug in exp_gen ([#606](https://github.com/microsoft/RD-Agent/issues/606)) ([f734dde](https://github.com/microsoft/RD-Agent/commit/f734dde0b0101e13f38151468c8ddf9e23af26ac))
|
||||
* add check when retrying gen model codes ([#699](https://github.com/microsoft/RD-Agent/issues/699)) ([3b82f15](https://github.com/microsoft/RD-Agent/commit/3b82f159474087902d3c6007d370e3282b549015))
|
||||
* add DSExperiment type check and directory validation in log proc… ([#535](https://github.com/microsoft/RD-Agent/issues/535)) ([f59b12c](https://github.com/microsoft/RD-Agent/commit/f59b12c9cc9afde82b74bc133797ff1396678627))
|
||||
* add ensemble test, change to "use cross-validation if possible" in workflow spec ([#634](https://github.com/microsoft/RD-Agent/issues/634)) ([acc97a8](https://github.com/microsoft/RD-Agent/commit/acc97a8217253497afedcfa829902b4432e1031e))
|
||||
* add force parameter for cache_with_pickle & using cache when get kaggle leaderboard ([#687](https://github.com/microsoft/RD-Agent/issues/687)) ([c8841e5](https://github.com/microsoft/RD-Agent/commit/c8841e590a925200859acba9fda4a17d4c3aa1c7))
|
||||
* add metric name check for valid scores ([#724](https://github.com/microsoft/RD-Agent/issues/724)) ([acc2ffb](https://github.com/microsoft/RD-Agent/commit/acc2ffbde4df3b53654559d14cd035ee6be6b35e))
|
||||
* add retry mechanism for GPU device check in DockerEnv ([#573](https://github.com/microsoft/RD-Agent/issues/573)) ([a780cfb](https://github.com/microsoft/RD-Agent/commit/a780cfb621dc487cc17072bfd4aedd7d581249ab))
|
||||
* add scores.csv checking in ensemble_test ([#567](https://github.com/microsoft/RD-Agent/issues/567)) ([01808b4](https://github.com/microsoft/RD-Agent/commit/01808b47c314d1daffacc0a65e0ab934a1c41d65))
|
||||
* add stdout context length setting and improve text shrinking logic ([#559](https://github.com/microsoft/RD-Agent/issues/559)) ([4ac26a6](https://github.com/microsoft/RD-Agent/commit/4ac26a65c1f18f7513480dd562566c8a96298aa7))
|
||||
* align components' name ([#701](https://github.com/microsoft/RD-Agent/issues/701)) ([295a114](https://github.com/microsoft/RD-Agent/commit/295a1148c53d00b716b2d540573a7f43e7e2d762))
|
||||
* auto continue small bug ([#598](https://github.com/microsoft/RD-Agent/issues/598)) ([75eaecf](https://github.com/microsoft/RD-Agent/commit/75eaecf36b9f70dfc2d7fedd35836acdb05f89d6))
|
||||
* avoid try-except in ensemble eval prompts ([#637](https://github.com/microsoft/RD-Agent/issues/637)) ([5c58d6e](https://github.com/microsoft/RD-Agent/commit/5c58d6e524ef848024578033ab6d47bc9b220822))
|
||||
* avoid warning for missing llama installation when not in use ([#509](https://github.com/microsoft/RD-Agent/issues/509)) ([5ec3422](https://github.com/microsoft/RD-Agent/commit/5ec342224c2c8c4cf591f1eae673e25b14218726))
|
||||
* change devault to default ([#688](https://github.com/microsoft/RD-Agent/issues/688)) ([7f401cd](https://github.com/microsoft/RD-Agent/commit/7f401cd1c3b333285acf6d6e57654f4b9f0cb6c5))
|
||||
* change ensemble test ([#622](https://github.com/microsoft/RD-Agent/issues/622)) ([5de3595](https://github.com/microsoft/RD-Agent/commit/5de35953ed0d3e2e1f4dff0e0522f2d6475079ec))
|
||||
* change summary info of log folder ([#552](https://github.com/microsoft/RD-Agent/issues/552)) ([0eb258d](https://github.com/microsoft/RD-Agent/commit/0eb258d734e9a1280a238b9a6f63eb33047ee0a7))
|
||||
* clarify an ambiguous explanation ([#705](https://github.com/microsoft/RD-Agent/issues/705)) ([5dbfc68](https://github.com/microsoft/RD-Agent/commit/5dbfc6859cbf6cc31932dae30cf05506108fc871))
|
||||
* clarify cross_validation ([#644](https://github.com/microsoft/RD-Agent/issues/644)) ([906993e](https://github.com/microsoft/RD-Agent/commit/906993ef6482f88131d1af46f5bc66a77034b549))
|
||||
* coder prompt & model test text ([#583](https://github.com/microsoft/RD-Agent/issues/583)) ([0a41227](https://github.com/microsoft/RD-Agent/commit/0a41227f267050feaeeb47ddd4d749643eb9f198))
|
||||
* correct the configuration inheritance relationship ([#671](https://github.com/microsoft/RD-Agent/issues/671)) ([30b1ff8](https://github.com/microsoft/RD-Agent/commit/30b1ff8e1ce59b741e0b81481962063014641c0b))
|
||||
* default emb model ([#702](https://github.com/microsoft/RD-Agent/issues/702)) ([4329a72](https://github.com/microsoft/RD-Agent/commit/4329a722832a201b3fa6f9d8f9d8d46f78110410))
|
||||
* direct_exp_gen to json_target_type in DSExpGen class ([#661](https://github.com/microsoft/RD-Agent/issues/661)) ([428b74a](https://github.com/microsoft/RD-Agent/commit/428b74a988157ea864ebb40e828bd9f67589c863))
|
||||
* docker error will trigger retry and data science runner loop set to 3 ([#602](https://github.com/microsoft/RD-Agent/issues/602)) ([ad785e0](https://github.com/microsoft/RD-Agent/commit/ad785e03d5db05d9191d5e772e184532835a787b))
|
||||
* ensure expected type ([#593](https://github.com/microsoft/RD-Agent/issues/593)) ([098a9a6](https://github.com/microsoft/RD-Agent/commit/098a9a6618f70fa8dd276b9014b9e7ba9621553b))
|
||||
* filter empty log traces in ds UI ([#533](https://github.com/microsoft/RD-Agent/issues/533)) ([1a2057c](https://github.com/microsoft/RD-Agent/commit/1a2057c9fc11edc4637f0baaa6dd226eb049c36e))
|
||||
* fix a bug in cross validation ([#618](https://github.com/microsoft/RD-Agent/issues/618)) ([05a4f10](https://github.com/microsoft/RD-Agent/commit/05a4f101e0b64b860ad03294619b2350004657e8))
|
||||
* fix a bug in ensemble test script ([#713](https://github.com/microsoft/RD-Agent/issues/713)) ([ad32100](https://github.com/microsoft/RD-Agent/commit/ad321000acbd9291d22fe03a9c60e57c70511c73))
|
||||
* fix a bug in initial tasks ([#635](https://github.com/microsoft/RD-Agent/issues/635)) ([edb552e](https://github.com/microsoft/RD-Agent/commit/edb552ed283119444f357fbd0b6170b2ad97712a))
|
||||
* fix a bug in kaggle conf ([#459](https://github.com/microsoft/RD-Agent/issues/459)) ([b4ed32b](https://github.com/microsoft/RD-Agent/commit/b4ed32b17ef07d8557450063765585a48d5fcd32))
|
||||
* fix a bug in progress_bar filter ([#712](https://github.com/microsoft/RD-Agent/issues/712)) ([ba5a84d](https://github.com/microsoft/RD-Agent/commit/ba5a84dee59c39cc2a8c0d428a82da1f899ce537))
|
||||
* fix a bug in proposal (add last loop's exception to last task desc) ([#596](https://github.com/microsoft/RD-Agent/issues/596)) ([419186f](https://github.com/microsoft/RD-Agent/commit/419186ffb985fe5a0aa0f7fe59c7a223e355492e))
|
||||
* fix a bug in regular expression exception processing ([#734](https://github.com/microsoft/RD-Agent/issues/734)) ([67d3702](https://github.com/microsoft/RD-Agent/commit/67d37027bbcd7294a5890a350fe16fe78e0dfa77))
|
||||
* fix a bug in threshold score display ([#592](https://github.com/microsoft/RD-Agent/issues/592)) ([0b0a2dc](https://github.com/microsoft/RD-Agent/commit/0b0a2dc512a5560a66464ad49de25d362d0dc17e))
|
||||
* fix a bug related to model_name in ensemble ([#692](https://github.com/microsoft/RD-Agent/issues/692)) ([c6ce473](https://github.com/microsoft/RD-Agent/commit/c6ce4733f32578298abe0b60f9d82611b793cc09))
|
||||
* fix a minor bug ([#694](https://github.com/microsoft/RD-Agent/issues/694)) ([1405d8d](https://github.com/microsoft/RD-Agent/commit/1405d8dafd99ecde6f3ba9dd76133d8830d03b47))
|
||||
* fix an error in model_coder prompt ([#690](https://github.com/microsoft/RD-Agent/issues/690)) ([4528826](https://github.com/microsoft/RD-Agent/commit/452882674e915dbd9e3399c26c70ce5bb86d012c))
|
||||
* fix combined_factors_df.pkl not loading in docker ([#697](https://github.com/microsoft/RD-Agent/issues/697)) ([3984b99](https://github.com/microsoft/RD-Agent/commit/3984b995aa74318b40de7712e100d4de5cc95b11))
|
||||
* fix docs build error ([#711](https://github.com/microsoft/RD-Agent/issues/711)) ([c9e1d32](https://github.com/microsoft/RD-Agent/commit/c9e1d32d6b63560350cc7cb799c3a908e2c04e42))
|
||||
* fix ExtendedSettingsConfigDict does not work ([#660](https://github.com/microsoft/RD-Agent/issues/660)) ([3a877f3](https://github.com/microsoft/RD-Agent/commit/3a877f383b908da8d027560714030b201946bb76))
|
||||
* fix kaggle templates path error ([#747](https://github.com/microsoft/RD-Agent/issues/747)) ([3b3f504](https://github.com/microsoft/RD-Agent/commit/3b3f5041514baf741fe2d4613fa651fb5d9c002d))
|
||||
* fix KeyError direct_exp_gen ([#735](https://github.com/microsoft/RD-Agent/issues/735)) ([7200682](https://github.com/microsoft/RD-Agent/commit/7200682ac4e60d3910c29a4f7c4a37b3d24e4224))
|
||||
* fix some bugs (ensemble output, HPO, model tuning) ([#648](https://github.com/microsoft/RD-Agent/issues/648)) ([818ee29](https://github.com/microsoft/RD-Agent/commit/818ee29f8e5d4765b9801463b85b42ee9516ec33))
|
||||
* fix some bugs in the ensemble component ([#595](https://github.com/microsoft/RD-Agent/issues/595)) ([c0990ab](https://github.com/microsoft/RD-Agent/commit/c0990abb06c73ae062d9a50f50cdfd6d04aded22))
|
||||
* fix some bugs in workflow unit test ([#624](https://github.com/microsoft/RD-Agent/issues/624)) ([f845dcc](https://github.com/microsoft/RD-Agent/commit/f845dcc0ee1b059b8b32485ad46bb90c7ae0fa78))
|
||||
* fix some description errors in direct_exp_gen ([#698](https://github.com/microsoft/RD-Agent/issues/698)) ([dfaacb6](https://github.com/microsoft/RD-Agent/commit/dfaacb6d06e5d5f55e950d7177570d1efebf958f))
|
||||
* fix some minor bugs and add AutoML & cross-validation ([#604](https://github.com/microsoft/RD-Agent/issues/604)) ([18c5ef2](https://github.com/microsoft/RD-Agent/commit/18c5ef268d40efe7bb9ee18aa0d250732bdda6fa))
|
||||
* fix submission file search and add TODO in env.py ([#544](https://github.com/microsoft/RD-Agent/issues/544)) ([54d930e](https://github.com/microsoft/RD-Agent/commit/54d930e91e629f0fc2f8bdd0d0d62fcad1e99a9c))
|
||||
* fix task return dict with wrong format ([#558](https://github.com/microsoft/RD-Agent/issues/558)) ([2008244](https://github.com/microsoft/RD-Agent/commit/20082440a249dd0e5a7026c2d98c9de0288dd400))
|
||||
* fix the errors in the coder and evaluator of the five components ([#576](https://github.com/microsoft/RD-Agent/issues/576)) ([c487f83](https://github.com/microsoft/RD-Agent/commit/c487f835b651cdc40b95bbbe4efcb9a617be9e40))
|
||||
* handle division by zero in percentage calculations ([#550](https://github.com/microsoft/RD-Agent/issues/550)) ([de16c91](https://github.com/microsoft/RD-Agent/commit/de16c915e1716ef8cee43ce41069ea1a09cf1f24))
|
||||
* handle invalid regex patterns in filter_progress_bar function ([#579](https://github.com/microsoft/RD-Agent/issues/579)) ([b0daee0](https://github.com/microsoft/RD-Agent/commit/b0daee0d90e193ca1d028e01c31ebf368af89601))
|
||||
* Handle ValueError when resolving relative path for uri ([#585](https://github.com/microsoft/RD-Agent/issues/585)) ([4c7765a](https://github.com/microsoft/RD-Agent/commit/4c7765a12bda5dcfd9af72b292853d9bc28c5baf))
|
||||
* include data information in cache key generation ([#566](https://github.com/microsoft/RD-Agent/issues/566)) ([26dda46](https://github.com/microsoft/RD-Agent/commit/26dda4682b7b643c164589057cb568a4d9e55e17))
|
||||
* keep some txt files ([#557](https://github.com/microsoft/RD-Agent/issues/557)) ([54aba85](https://github.com/microsoft/RD-Agent/commit/54aba851c9fa194e318d37700307df59e06c6c84))
|
||||
* mle_score save problem ([#674](https://github.com/microsoft/RD-Agent/issues/674)) ([ca2e478](https://github.com/microsoft/RD-Agent/commit/ca2e478cf25c2c8511d5f027e32f8a98fc8e3a07))
|
||||
* move docker timeout message to __run() ([#620](https://github.com/microsoft/RD-Agent/issues/620)) ([585f4f9](https://github.com/microsoft/RD-Agent/commit/585f4f96e09f70d00eb397c10bf49c09973111df))
|
||||
* move mlebench check into runner ([#556](https://github.com/microsoft/RD-Agent/issues/556)) ([b0f7965](https://github.com/microsoft/RD-Agent/commit/b0f7965f650638273710302efee2e5da037368a2))
|
||||
* move next_component_required logic to DSTrace class and accurate implement ([#612](https://github.com/microsoft/RD-Agent/issues/612)) ([c20d311](https://github.com/microsoft/RD-Agent/commit/c20d311792f33b2ccccb466c6ec3155ff8be3213))
|
||||
* patching weird azure deployment ([#494](https://github.com/microsoft/RD-Agent/issues/494)) ([89c50ae](https://github.com/microsoft/RD-Agent/commit/89c50aee2ec8bfd1cb23767ddf7dcdd023daac8b))
|
||||
* qlib and other scenario bugs ([#636](https://github.com/microsoft/RD-Agent/issues/636)) ([98de31d](https://github.com/microsoft/RD-Agent/commit/98de31d4e577c8c450c9694f73a755c19af571f7))
|
||||
* refine prompt to generate the most simple task in init stage ([#546](https://github.com/microsoft/RD-Agent/issues/546)) ([9d6feed](https://github.com/microsoft/RD-Agent/commit/9d6feed28ce034db48482d8d9741ef8c72f4bddc))
|
||||
* replace API call with build_cls_from_json_with_retry function ([#548](https://github.com/microsoft/RD-Agent/issues/548)) ([eb72a47](https://github.com/microsoft/RD-Agent/commit/eb72a47fbf9c88dacea9691b8d7e92610492d190))
|
||||
* replace func "len()" in ensemble test code to support various data type ([#739](https://github.com/microsoft/RD-Agent/issues/739)) ([ab9c7b9](https://github.com/microsoft/RD-Agent/commit/ab9c7b955f78c5de7ec08a6c1a012a76badbdd0e))
|
||||
* return 1D embedding if create_embedding receive a string input ([#670](https://github.com/microsoft/RD-Agent/issues/670)) ([4a9c318](https://github.com/microsoft/RD-Agent/commit/4a9c3180ae4a4b043b1b4a89f51ee69cb6843142))
|
||||
* rich.print error when some control char in output ([#684](https://github.com/microsoft/RD-Agent/issues/684)) ([ec0cb2a](https://github.com/microsoft/RD-Agent/commit/ec0cb2a032824023dcd04a3acc93202471d1f90a))
|
||||
* Runnable on first complete & Rename method to next_incomplete_component for clarity ([#615](https://github.com/microsoft/RD-Agent/issues/615)) ([93d9f63](https://github.com/microsoft/RD-Agent/commit/93d9f63369a78f78e1a67ab548923bb994d1d3b4))
|
||||
* runner COSTEER evaluator ([#693](https://github.com/microsoft/RD-Agent/issues/693)) ([6a379ec](https://github.com/microsoft/RD-Agent/commit/6a379ec9b84d4e4944f1e412347aae4f5a93d476))
|
||||
* save only one mle_score pkl for a running exp ([#675](https://github.com/microsoft/RD-Agent/issues/675)) ([f87ab67](https://github.com/microsoft/RD-Agent/commit/f87ab676b73cce82bd9f997ac779e31c571b53c4))
|
||||
* Set default value for 'entry' parameter in Env.run method ([#643](https://github.com/microsoft/RD-Agent/issues/643)) ([e50d242](https://github.com/microsoft/RD-Agent/commit/e50d2424b849e4181d6ca02e9cace90236665924))
|
||||
* sort file name for cache reproduction ([#588](https://github.com/microsoft/RD-Agent/issues/588)) ([7158410](https://github.com/microsoft/RD-Agent/commit/7158410fbfdd84052f9a69cf1e04e09ac07ca598))
|
||||
* sota comparison logic ([#608](https://github.com/microsoft/RD-Agent/issues/608)) ([3575372](https://github.com/microsoft/RD-Agent/commit/35753722c0800d62855faeab996d513e62cfe7de))
|
||||
* target json type & round ([#662](https://github.com/microsoft/RD-Agent/issues/662)) ([58cb58f](https://github.com/microsoft/RD-Agent/commit/58cb58f966a1db26f5ea9662a54ba12bc921ee24))
|
||||
* templates bug ([#456](https://github.com/microsoft/RD-Agent/issues/456)) ([434a868](https://github.com/microsoft/RD-Agent/commit/434a8687eeda77e27b4938fb19694c15858ee446))
|
||||
* trace summary df showing in dsapp ([#551](https://github.com/microsoft/RD-Agent/issues/551)) ([177096d](https://github.com/microsoft/RD-Agent/commit/177096d55fecb8c7dab9650ef8f5a31024cd4c1c))
|
||||
* unzip kaggle data ([#464](https://github.com/microsoft/RD-Agent/issues/464)) ([3a9fc8e](https://github.com/microsoft/RD-Agent/commit/3a9fc8e73337d3757267b6f4482499499a1b6792))
|
||||
|
||||
## [0.3.0](https://github.com/microsoft/RD-Agent/compare/v0.2.1...v0.3.0) (2024-10-21)
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,50 @@
|
||||
# Contributing to RD-Agent
|
||||
|
||||
We welcome contributions and suggestions to improve RD-Agent. Whether it's solving an issue, addressing a bug, enhancing documentation, or even correcting a typo, every contribution is valuable and helps improve the project.
|
||||
|
||||
## Getting Started
|
||||
|
||||
To get started, you can explore the issues list or search for `TODO:` comments in the codebase by running the command:
|
||||
```sh
|
||||
grep -r "TODO:"
|
||||
```
|
||||
|
||||
## How to Contribute
|
||||
|
||||
1. **Fork the Repository**: Create a fork of the repository on GitHub.
|
||||
2. **Clone the Repository**: Clone your forked repository to your local machine.
|
||||
```sh
|
||||
git clone https://github.com/your-username/RD-Agent.git
|
||||
```
|
||||
3. **Create a Branch**: Create a new branch for your changes.
|
||||
```sh
|
||||
git checkout -b feature/your-feature-name
|
||||
```
|
||||
4. **Make Changes**: Make your changes to the codebase.
|
||||
5. **Commit Changes**: Commit your changes with a descriptive commit message.
|
||||
```sh
|
||||
git commit -m "Description of your changes"
|
||||
```
|
||||
6. **Push Changes**: Push your changes to your forked repository.
|
||||
```sh
|
||||
git push origin feature/your-feature-name
|
||||
```
|
||||
7. **Ensure CI Passes**: Make sure your code passes the automatic CI checks on GitHub.
|
||||
8. **Create a Pull Request**: Create a pull request from your forked repository to the main repository.
|
||||
|
||||
## Code of Conduct
|
||||
|
||||
Please adhere to the [Code of Conduct](CODE_OF_CONDUCT.md) in all your interactions with the project.
|
||||
|
||||
## Reporting Issues
|
||||
|
||||
If you encounter any issues or have suggestions for improvements, please open an issue on GitHub.
|
||||
|
||||
## Guidelines
|
||||
|
||||
- Ensure your code follows the project's coding standards.
|
||||
- Write clear and concise commit messages.
|
||||
- Update documentation as needed.
|
||||
- Test your changes thoroughly before submitting a pull request.
|
||||
|
||||
Thank you for contributing to RD-Agent!
|
||||
@@ -68,6 +68,7 @@ init-qlib-env:
|
||||
|
||||
dev:
|
||||
$(PIPRUN) pip install -e .[docs,lint,package,test] -c $(CONSTRAINTS_FILE)
|
||||
$(PIPRUN) pip install -U kaggle
|
||||
if [ "$(CI)" != "true" ] && command -v pre-commit > /dev/null 2>&1; then pre-commit install --hook-type pre-push; fi
|
||||
|
||||
# Generate constraints for current Python version.
|
||||
@@ -91,7 +92,7 @@ isort:
|
||||
# First deal with the core folder, and then gradually increase the scope of detection,
|
||||
# and eventually realize the detection of the complete project.
|
||||
mypy:
|
||||
$(PIPRUN) python -m mypy rdagent/core # --exclude rdagent/scripts,git_ignore_folder
|
||||
$(PIPRUN) python -m mypy rdagent/core
|
||||
|
||||
# Check lint with ruff.
|
||||
# First deal with the core folder, and then gradually increase the scope of detection,
|
||||
|
||||
@@ -1,7 +1,11 @@
|
||||
<h4 align="center">
|
||||
<img src="docs/_static/logo.png" alt="RA-Agent logo" style="width:70%; ">
|
||||
|
||||
<a href="https://rdagent.azurewebsites.net" target="_blank">🖥️ Live Demo</a> | <a href="https://rdagent.azurewebsites.net/factor_loop" target="_blank">🎥 Demo Video</a> <a href="https://www.youtube.com/watch?v=JJ4JYO3HscM&list=PLALmKB0_N3_i52fhUmPQiL4jsO354uopR" target="_blank">▶️YouTube</a> | <a href="https://rdagent.readthedocs.io/en/latest/index.html" target="_blank">📖 Documentation</a> | <a href="#-paperwork-list"> 📃 Papers </a>
|
||||
<a href="https://rdagent.azurewebsites.net" target="_blank">🖥️ Live Demo</a> |
|
||||
<a href="https://rdagent.azurewebsites.net/factor_loop" target="_blank">🎥 Demo Video</a> <a href="https://www.youtube.com/watch?v=JJ4JYO3HscM&list=PLALmKB0_N3_i52fhUmPQiL4jsO354uopR" target="_blank">▶️YouTube</a> |
|
||||
<a href="https://rdagent.readthedocs.io/en/latest/index.html" target="_blank">📖 Documentation</a> |
|
||||
<a href="https://aka.ms/RD-Agent-Tech-Report" target="_blank">📄 Tech Report</a> |
|
||||
<a href="#-paperwork-list"> 📃 Papers </a>
|
||||
</h3>
|
||||
|
||||
|
||||
@@ -19,31 +23,82 @@
|
||||
[](http://mypy-lang.org/)
|
||||
[](https://github.com/astral-sh/ruff)
|
||||
[](https://discord.gg/ybQ97B6Jjy)
|
||||
[](https://rdagent.readthedocs.io/en/latest/?badge=latest)
|
||||
[](https://github.com/microsoft/RD-Agent/actions/workflows/readthedocs-preview.yml) <!-- this badge is too long, please place it in the last one to make it pretty -->
|
||||
[](https://arxiv.org/abs/2505.14738)
|
||||
|
||||
|
||||
|
||||
# 🏆 The Best Machine Learning Engineering Agent!
|
||||
|
||||
[MLE-bench](https://github.com/openai/mle-bench) is a comprehensive benchmark evaluating the performance of AI agents on machine learning engineering tasks. Utilizing datasets from 75 Kaggle competitions, MLE-bench provides robust assessments of AI systems' capabilities in real-world ML engineering scenarios.
|
||||
|
||||
R&D-Agent currently leads as the top-performing machine learning engineering agent on MLE-bench:
|
||||
|
||||
| Agent | Low == Lite (%) | Medium (%) | High (%) | All (%) |
|
||||
|---------|--------|-----------|---------|----------|
|
||||
| R&D-Agent o1-preview | 48.18 ± 2.49 | 8.95 ± 2.36 | 18.67 ± 2.98 | 22.4 ± 1.1 |
|
||||
| R&D-Agent o3(R)+GPT-4.1(D) | 51.52 ± 6.21 | 7.89 ± 3.33 | 16.67 ± 3.65 | 22.45 ± 2.45 |
|
||||
| AIDE o1-preview | 34.3 ± 2.4 | 8.8 ± 1.1 | 10.0 ± 1.9 | 16.9 ± 1.1 |
|
||||
|
||||
**Notes:**
|
||||
- **O3(R)+GPT-4.1(D)**: This version is designed to both reduce average time per loop and leverage a cost-effective combination of backend LLMs by seamlessly integrating Research Agent (o3) with Development Agent (GPT-4.1).
|
||||
- **AIDE o1-preview**: Represents the previously best public result on MLE-bench as reported in the original MLE-bench paper.
|
||||
- Average and standard deviation results for R&D-Agent o1-preview is based on a independent of 5 seeds and for R&D-Agent o3(R)+GPT-4.1(D) is based on 6 seeds.
|
||||
- According to MLE-Bench, the 75 competitions are categorized into three levels of complexity: **Low==Lite** if we estimate that an experienced ML engineer can produce a sensible solution in under 2 hours, excluding the time taken to train any models; **Medium** if it takes between 2 and 10 hours; and **High** if it takes more than 10 hours.
|
||||
|
||||
You can inspect the detailed runs of the above results online.
|
||||
- [R&D-Agent o1-preview detailed runs](https://aka.ms/RD-Agent_MLE-Bench_O1-preview)
|
||||
- [R&D-Agent o3(R)+GPT-4.1(D) detailed runs](https://aka.ms/RD-Agent_MLE-Bench_O3_GPT41)
|
||||
|
||||
For running R&D-Agent on MLE-bench, refer to **[MLE-bench Guide: Running ML Engineering via MLE-bench](https://rdagent.readthedocs.io/en/latest/scens/data_science.html)**
|
||||
|
||||
# 🥇 The First Data-Centric Quant Multi-Agent Framework!
|
||||
|
||||
R&D-Agent for Quantitative Finance, in short **RD-Agent(Q)**, is the first data-centric, multi-agent framework designed to automate the full-stack research and development of quantitative strategies via coordinated factor-model co-optimization.
|
||||
|
||||

|
||||
|
||||
Extensive experiments in real stock markets show that, at a cost under $10, RD-Agent(Q) achieves approximately 2× higher ARR than benchmark factor libraries while using over 70% fewer factors. It also surpasses state-of-the-art deep time-series models under smaller resource budgets. Its alternating factor–model optimization further delivers excellent trade-off between predictive accuracy and strategy robustness.
|
||||
|
||||
You can learn more details about **RD-Agent(Q)** through the [paper](https://arxiv.org/abs/2505.15155) and reproduce it through the [documentation](https://rdagent.readthedocs.io/en/latest/scens/quant_agent_fin.html).
|
||||
|
||||
# 📰 News
|
||||
| 🗞️ News | 📝 Description |
|
||||
| -- | ------ |
|
||||
| Official WeChat group release | We created a WeChat group, welcome to join! (🗪[QR Code](docs/WeChat_QR_code.jpg)) |
|
||||
| -- | ------ |
|
||||
| [Technical Report Release](#overall-technical-report) | Overall framework description and results on MLE-bench |
|
||||
| [R&D-Agent-Quant Release](#deep-application-in-diverse-scenarios) | Apply R&D-Agent to quant trading |
|
||||
| MLE-Bench Results Released | R&D-Agent currently leads as the [top-performing machine learning engineering agent](#-the-best-machine-learning-engineering-agent) on MLE-bench |
|
||||
| Support LiteLLM Backend | We now fully support **[LiteLLM](https://github.com/BerriAI/litellm)** as a backend for integration with multiple LLM providers. |
|
||||
| General Data Science Agent | [Data Science Agent](https://rdagent.readthedocs.io/en/latest/scens/data_science.html) |
|
||||
| Kaggle Scenario release | We release **[Kaggle Agent](https://rdagent.readthedocs.io/en/latest/scens/data_science.html)**, try the new features! |
|
||||
| Official WeChat group release | We created a WeChat group, welcome to join! (🗪[QR Code](https://github.com/microsoft/RD-Agent/issues/880)) |
|
||||
| Official Discord release | We launch our first chatting channel in Discord (🗪[](https://discord.gg/ybQ97B6Jjy)) |
|
||||
| First release | **RDAgent** is released on GitHub |
|
||||
| First release | **R&D-Agent** is released on GitHub |
|
||||
|
||||
|
||||
|
||||
# Data Science Agent Preview
|
||||
Check out our demo video showcasing the current progress of our Data Science Agent under development:
|
||||
|
||||
https://github.com/user-attachments/assets/3eccbecb-34a4-4c81-bce4-d3f8862f7305
|
||||
|
||||
# 🌟 Introduction
|
||||
<div align="center">
|
||||
<img src="docs/_static/scen.png" alt="Our focused scenario" style="width:80%; ">
|
||||
</div>
|
||||
|
||||
RDAgent aims to automate the most critical and valuable aspects of the industrial R&D process, and we begin with focusing on the data-driven scenarios to streamline the development of models and data.
|
||||
R&D-Agent aims to automate the most critical and valuable aspects of the industrial R&D process, and we begin with focusing on the data-driven scenarios to streamline the development of models and data.
|
||||
Methodologically, we have identified a framework with two key components: 'R' for proposing new ideas and 'D' for implementing them.
|
||||
We believe that the automatic evolution of R&D will lead to solutions of significant industrial value.
|
||||
|
||||
|
||||
<!-- Tag Cloud -->
|
||||
R&D is a very general scenario. The advent of RDAgent can be your
|
||||
R&D is a very general scenario. The advent of R&D-Agent can be your
|
||||
- 💰 **Automatic Quant Factory** ([🎥Demo Video](https://rdagent.azurewebsites.net/factor_loop)|[▶️YouTube](https://www.youtube.com/watch?v=X4DK2QZKaKY&t=6s))
|
||||
- 🤖 **Data Mining Agent:** Iteratively proposing data & models ([🎥Demo Video 1](https://rdagent.azurewebsites.net/model_loop)|[▶️YouTube](https://www.youtube.com/watch?v=dm0dWL49Bc0&t=104s)) ([🎥Demo Video 2](https://rdagent.azurewebsites.net/dmm)|[▶️YouTube](https://www.youtube.com/watch?v=VIaSTZuoZg4)) and implementing them by gaining knowledge from data.
|
||||
- 🦾 **Research Copilot:** Auto read research papers ([🎥Demo Video](https://rdagent.azurewebsites.net/report_model)|[▶️YouTube](https://www.youtube.com/watch?v=BiA2SfdKQ7o)) / financial reports ([🎥Demo Video](https://rdagent.azurewebsites.net/report_factor)|[▶️YouTube](https://www.youtube.com/watch?v=ECLTXVcSx-c)) and implement model structures or building datasets.
|
||||
- 🤖 **Kaggle Agent:** Auto Model Tuning and Feature Engineering([🎥Demo Video Coming Soon...]()) and implementing them to achieve more in competitions.
|
||||
- ...
|
||||
|
||||
You can click the links above to view the demo. We're continuously adding more methods and scenarios to the project to enhance your R&D processes and boost productivity.
|
||||
@@ -63,6 +118,7 @@ You can try above demos by running the following command:
|
||||
|
||||
### 🐳 Docker installation.
|
||||
Users must ensure Docker is installed before attempting most scenarios. Please refer to the [official 🐳Docker page](https://docs.docker.com/engine/install/) for installation instructions.
|
||||
Ensure the current user can run Docker commands **without using sudo**. You can verify this by executing `docker run hello-world`.
|
||||
|
||||
### 🐍 Create a Conda Environment
|
||||
- Create a new conda environment with Python (3.10 and 3.11 are well-tested in our CI):
|
||||
@@ -74,26 +130,77 @@ Users must ensure Docker is installed before attempting most scenarios. Please r
|
||||
conda activate rdagent
|
||||
```
|
||||
|
||||
### 🛠️ Install the RDAgent
|
||||
- You can directly install the RDAgent package from PyPI:
|
||||
### 🛠️ Install the R&D-Agent
|
||||
- You can directly install the R&D-Agent package from PyPI:
|
||||
```sh
|
||||
pip install rdagent
|
||||
```
|
||||
|
||||
### 💊 Health check
|
||||
- rdagent provides a health check that currently checks two things.
|
||||
- whether the docker installation was successful.
|
||||
- whether the default port used by the [rdagent ui](https://github.com/microsoft/RD-Agent?tab=readme-ov-file#%EF%B8%8F-monitor-the-application-results) is occupied.
|
||||
```sh
|
||||
rdagent health_check
|
||||
```
|
||||
|
||||
|
||||
### ⚙️ Configuration
|
||||
- You have to config your GPT model in the `.env`
|
||||
- The demos requires following ability:
|
||||
- ChatCompletion
|
||||
- json_mode
|
||||
- embedding query
|
||||
|
||||
You can set your Chat Model and Embedding Model in the following ways:
|
||||
|
||||
- **Using LiteLLM (Default)**: We now support LiteLLM as a backend for integration with multiple LLM providers. You can configure in two ways:
|
||||
|
||||
**Option 1: Unified API base for both models**
|
||||
```bash
|
||||
cat << EOF > .env
|
||||
OPENAI_API_KEY=<your_api_key>
|
||||
# EMBEDDING_MODEL=text-embedding-3-small
|
||||
CHAT_MODEL=gpt-4-turbo
|
||||
EOF
|
||||
# Set to any model supported by LiteLLM.
|
||||
CHAT_MODEL=gpt-4o
|
||||
EMBEDDING_MODEL=text-embedding-3-small
|
||||
# Configure unified API base
|
||||
OPENAI_API_BASE=<your_unified_api_base>
|
||||
OPENAI_API_KEY=<replace_with_your_openai_api_key>
|
||||
```
|
||||
|
||||
**Option 2: Separate API bases for Chat and Embedding models**
|
||||
```bash
|
||||
cat << EOF > .env
|
||||
# Set to any model supported by LiteLLM.
|
||||
# Configure separate API bases for chat and embedding
|
||||
|
||||
# CHAT MODEL:
|
||||
CHAT_MODEL=gpt-4o
|
||||
OPENAI_API_BASE=<your_chat_api_base>
|
||||
OPENAI_API_KEY=<replace_with_your_openai_api_key>
|
||||
|
||||
# EMBEDDING MODEL:
|
||||
# TAKE siliconflow as an example, you can use other providers.
|
||||
# Note: embedding requires litellm_proxy prefix
|
||||
EMBEDDING_MODEL=litellm_proxy/BAAI/bge-large-en-v1.5
|
||||
LITELLM_PROXY_API_KEY=<replace_with_your_siliconflow_api_key>
|
||||
LITELLM_PROXY_API_BASE=https://api.siliconflow.cn/v1
|
||||
```
|
||||
|
||||
Notice: If you are using reasoning models that include thought processes in their responses (such as \<think> tags), you need to set the following environment variable:
|
||||
```bash
|
||||
REASONING_THINK_RM=True
|
||||
```
|
||||
|
||||
- You can also use a deprecated backend if you only use `OpenAI API` or `Azure OpenAI` directly. For this deprecated setting and more configuration information, please refer to the [documentation](https://rdagent.readthedocs.io/en/latest/installation_and_configuration.html).
|
||||
|
||||
### 🚀 Run the Application
|
||||
|
||||
The **[🖥️ Live Demo](https://rdagent.azurewebsites.net/)** is implemented by the following commands(each item represents one demo, you can select the one you prefer):
|
||||
|
||||
- Run the **Automated Quantitative Trading & Iterative Factors Model Joint Evolution**: [Qlib](http://github.com/microsoft/qlib) self-loop factor & model proposal and implementation application
|
||||
```sh
|
||||
rdagent fin_quant
|
||||
```
|
||||
|
||||
- Run the **Automated Quantitative Trading & Iterative Factors Evolution**: [Qlib](http://github.com/microsoft/qlib) self-loop factor proposal and implementation application
|
||||
```sh
|
||||
rdagent fin_factor
|
||||
@@ -104,19 +211,6 @@ The **[🖥️ Live Demo](https://rdagent.azurewebsites.net/)** is implemented b
|
||||
rdagent fin_model
|
||||
```
|
||||
|
||||
- Run the **Automated Medical Prediction Model Evolution**: Medical self-loop model proposal and implementation application
|
||||
>(1) Apply for an account at [PhysioNet](https://physionet.org/). <br /> (2) Request access to FIDDLE preprocessed data: [FIDDLE Dataset](https://physionet.org/content/mimic-eicu-fiddle-feature/1.0.0/). <br />
|
||||
(3) Place your username and password in `.env`.
|
||||
```bash
|
||||
cat << EOF >> .env
|
||||
DM_USERNAME=<your_username>
|
||||
DM_PASSWORD=<your_password>
|
||||
EOF
|
||||
```
|
||||
```sh
|
||||
rdagent med_model
|
||||
```
|
||||
|
||||
- Run the **Automated Quantitative Trading & Factors Extraction from Financial Reports**: Run the [Qlib](http://github.com/microsoft/qlib) factor extraction and implementation application based on financial reports
|
||||
```sh
|
||||
# 1. Generally, you can run this scenario using the following command:
|
||||
@@ -137,15 +231,49 @@ The **[🖥️ Live Demo](https://rdagent.azurewebsites.net/)** is implemented b
|
||||
rdagent general_model "https://arxiv.org/pdf/2210.09789"
|
||||
```
|
||||
|
||||
- Run the **Automated Kaggle Model Tuning & Feature Engineering**: self-loop model proposal and feature engineering implementation application <br />
|
||||
> Using **sf-crime** *(San Francisco Crime Classification)* as an example. <br />
|
||||
> 1. Register and login on the [Kaggle](https://www.kaggle.com/) website. <br />
|
||||
> 2. Configuring the Kaggle API. <br />
|
||||
> (1) Click on the avatar (usually in the top right corner of the page) -> `Settings` -> `Create New Token`, A file called `kaggle.json` will be downloaded. <br />
|
||||
> (2) Move `kaggle.json` to `~/.config/kaggle/` <br />
|
||||
> (3) Modify the permissions of the kaggle.json file. Reference command: `chmod 600 ~/.config/kaggle/kaggle.json` <br />
|
||||
> 3. Join the competition: Click `Join the competition` -> `I Understand and Accept` at the bottom of the [competition details page](https://www.kaggle.com/competitions/sf-crime/data).
|
||||
```bash
|
||||
# Generally, you can run the Kaggle competition program with the following command:
|
||||
rdagent data_science --competition <your competition name>
|
||||
|
||||
# Specifically, you need to create a folder for storing competition files (e.g., competition description file, competition datasets, etc.), and configure the path to the folder in your environment. In addition, you need to use chromedriver when you download the competition descriptors, which you can follow for this specific example:
|
||||
|
||||
# 1. Install chromedriver.
|
||||
|
||||
# 2. Add the competition description file path to the `.env` file.
|
||||
mkdir -p ./git_ignore_folder/kaggle_data
|
||||
dotenv set DS_LOCAL_DATA_PATH "$(pwd)/git_ignore_folder/kaggle_data"
|
||||
dotenv set DS_IF_USING_MLE_DATA True
|
||||
|
||||
# 3. run the application
|
||||
rdagent data_science --competition sf-crime
|
||||
```
|
||||
|
||||
### 🖥️ Monitor the Application Results
|
||||
- You can serve our demo app to monitor the RD loop by running the following command:
|
||||
- You can run the following command for our demo program to see the run logs.
|
||||
|
||||
```sh
|
||||
rdagent ui --port 80 --log_dir <your log folder like "log/">
|
||||
rdagent ui --port 19899 --log_dir <your log folder like "log/">
|
||||
```
|
||||
|
||||
**Note:** Although port 19899 is not commonly used, but before you run this demo, you need to check if port 19899 is occupied. If it is, please change it to another port that is not occupied.
|
||||
|
||||
You can check if a port is occupied by running the following command.
|
||||
|
||||
```sh
|
||||
rdagent health_check
|
||||
```
|
||||
|
||||
# 🏭 Scenarios
|
||||
|
||||
We have applied RD-Agent to multiple valuable data-driven industrial scenarios.
|
||||
We have applied R&D-Agent to multiple valuable data-driven industrial scenarios.
|
||||
|
||||
|
||||
## 🎯 Goal: Agent for Data-driven R&D
|
||||
@@ -170,15 +298,13 @@ The supported scenarios are listed below:
|
||||
| -- | -- | -- |
|
||||
| **💹 Finance** | 🤖 [Iteratively Proposing Ideas & Evolving](https://rdagent.azurewebsites.net/model_loop)[▶️YouTube](https://www.youtube.com/watch?v=dm0dWL49Bc0&t=104s) | 🤖 [Iteratively Proposing Ideas & Evolving](https://rdagent.azurewebsites.net/factor_loop) [▶️YouTube](https://www.youtube.com/watch?v=X4DK2QZKaKY&t=6s) <br/> 🦾 [Auto reports reading & implementation](https://rdagent.azurewebsites.net/report_factor)[▶️YouTube](https://www.youtube.com/watch?v=ECLTXVcSx-c) |
|
||||
| **🩺 Medical** | 🤖 [Iteratively Proposing Ideas & Evolving](https://rdagent.azurewebsites.net/dmm)[▶️YouTube](https://www.youtube.com/watch?v=VIaSTZuoZg4) | - |
|
||||
| **🏭 General** | 🦾 [Auto paper reading & implementation](https://rdagent.azurewebsites.net/report_model)[▶️YouTube](https://www.youtube.com/watch?v=BiA2SfdKQ7o) | - |
|
||||
| **🏭 General** | 🦾 [Auto paper reading & implementation](https://rdagent.azurewebsites.net/report_model)[▶️YouTube](https://www.youtube.com/watch?v=BiA2SfdKQ7o) <br/> 🤖 Auto Kaggle Model Tuning | 🤖Auto Kaggle feature Engineering |
|
||||
|
||||
- **[RoadMap](https://rdagent.readthedocs.io/en/latest/scens/data_science.html#roadmap)**: Currently, we are working hard to add new features to the Kaggle scenario.
|
||||
|
||||
Different scenarios vary in entrance and configuration. Please check the detailed setup tutorial in the scenarios documents.
|
||||
|
||||
Here is a gallery of [successful explorations](https://github.com/SunsetWolf/rdagent_resource/releases/download/demo_traces/demo_traces.zip) (5 traces showed in **[🖥️ Live Demo](https://rdagent.azurewebsites.net/)**). You can download and view the execution trace using the command below:
|
||||
|
||||
```bash
|
||||
rdagent ui --port 80 --log_dir ./demo_traces
|
||||
```
|
||||
Here is a gallery of [successful explorations](https://github.com/SunsetWolf/rdagent_resource/releases/download/demo_traces/demo_traces.zip) (5 traces showed in **[🖥️ Live Demo](https://rdagent.azurewebsites.net/)**). You can download and view the execution trace using [this command](https://github.com/microsoft/RD-Agent?tab=readme-ov-file#%EF%B8%8F-monitor-the-application-results) from the documentation.
|
||||
|
||||
Please refer to **[📖readthedocs_scen](https://rdagent.readthedocs.io/en/latest/scens/catalog.html)** for more details of the scenarios.
|
||||
|
||||
@@ -204,6 +330,21 @@ More documents can be found in the **[📖 readthedocs](https://rdagent.readthed
|
||||
|
||||
# 📃 Paper/Work list
|
||||
|
||||
## Overall Technical Report
|
||||
- [R&D-Agent: Automating Data-Driven AI Solution Building Through LLM-Powered Automated Research, Development, and Evolution](https://arxiv.org/abs/2505.14738)
|
||||
```BibTeX
|
||||
@misc{yang2024rdagent,
|
||||
title={R\&D-Agent: Automating Data-Driven AI Solution Building Through LLM-Powered Automated Research, Development, and Evolution},
|
||||
author={Xu Yang and Xiao Yang and Shikai Fang and Bowen Xian and Yuante Li and Jian Wang and Minrui Xu and Haoran Pan and Xinpeng Hong and Weiqing Liu and Yelong Shen and Weizhu Chen and Jiang Bian},
|
||||
year={2025},
|
||||
eprint={2505.14738},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={cs.AI},
|
||||
url={https://arxiv.org/abs/2505.14738}
|
||||
}
|
||||
```
|
||||

|
||||
|
||||
## 📊 Benchmark
|
||||
- [Towards Data-Centric Automatic R&D](https://arxiv.org/abs/2404.11276)
|
||||
```BibTeX
|
||||
@@ -241,12 +382,31 @@ For more detail, please refer to our **[🖥️ Live Demo page](https://rdagent.
|
||||
```
|
||||

|
||||
|
||||
## Deep Application in Diverse Scenarios
|
||||
|
||||
- [R&D-Agent-Quant: A Multi-Agent Framework for Data-Centric Factors and Model Joint Optimization](https://arxiv.org/abs/2505.15155)
|
||||
```BibTeX
|
||||
@misc{li2025rdagentquant,
|
||||
title={R\&D-Agent-Quant: A Multi-Agent Framework for Data-Centric Factors and Model Joint Optimization},
|
||||
author={Yuante Li and Xu Yang and Xiao Yang and Minrui Xu and Xisen Wang and Weiqing Liu and Jiang Bian},
|
||||
year={2025},
|
||||
eprint={2505.15155},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={cs.AI}
|
||||
}
|
||||
```
|
||||

|
||||
|
||||
|
||||
# 🤝 Contributing
|
||||
|
||||
We welcome contributions and suggestions to improve R&D-Agent. Please refer to the [Contributing Guide](CONTRIBUTING.md) for more details on how to contribute.
|
||||
|
||||
Before submitting a pull request, ensure that your code passes the automatic CI checks.
|
||||
|
||||
## 📝 Guidelines
|
||||
This project welcomes contributions and suggestions.
|
||||
Contributing to this project is straightforward and rewarding. Whether it's solving an issue, addressing a bug, enhancing documentation, or even correcting a typo, every contribution is valuable and helps improve RDAgent.
|
||||
Contributing to this project is straightforward and rewarding. Whether it's solving an issue, addressing a bug, enhancing documentation, or even correcting a typo, every contribution is valuable and helps improve R&D-Agent.
|
||||
|
||||
To get started, you can explore the issues list, or search for `TODO:` comments in the codebase by running the command `grep -r "TODO:"`.
|
||||
|
||||
@@ -256,7 +416,7 @@ To get started, you can explore the issues list, or search for `TODO:` comments
|
||||
<img src="https://contrib.rocks/image?repo=microsoft/RD-Agent&max=100&columns=15" />
|
||||
</a>
|
||||
|
||||
Before we released RD-Agent as an open-source project on GitHub, it was an internal project within our group. Unfortunately, the internal commit history was not preserved when we removed some confidential code. As a result, some contributions from our group members, including Haotian Chen, Wenjun Feng, Haoxue Wang, Zeqi Ye, Xinjie Shen, and Jinhui Li, were not included in the public commits.
|
||||
Before we released R&D-Agent as an open-source project on GitHub, it was an internal project within our group. Unfortunately, the internal commit history was not preserved when we removed some confidential code. As a result, some contributions from our group members, including Haotian Chen, Wenjun Feng, Haoxue Wang, Zeqi Ye, Xinjie Shen, and Jinhui Li, were not included in the public commits.
|
||||
|
||||
# ⚖️ Legal disclaimer
|
||||
<p style="line-height: 1; font-style: italic;">The RD-agent is provided “as is”, without warranty of any kind, express or implied, including but not limited to the warranties of merchantability, fitness for a particular purpose and noninfringement. The RD-agent is aimed to facilitate research and development process in the financial industry and not ready-to-use for any financial investment or advice. Users shall independently assess and test the risks of the RD-agent in a specific use scenario, ensure the responsible use of AI technology, including but not limited to developing and integrating risk mitigation measures, and comply with all applicable laws and regulations in all applicable jurisdictions. The RD-agent does not provide financial opinions or reflect the opinions of Microsoft, nor is it designed to replace the role of qualified financial professionals in formulating, assessing, and approving finance products. The inputs and outputs of the RD-agent belong to the users and users shall assume all liability under any theory of liability, whether in contract, torts, regulatory, negligence, products liability, or otherwise, associated with use of the RD-agent and any inputs and outputs thereof.</p>
|
||||
|
||||
+7
-242
@@ -1,243 +1,8 @@
|
||||
aiohttp==3.9.1
|
||||
aiosignal==1.3.1
|
||||
alabaster==0.7.13
|
||||
annotated-types==0.6.0
|
||||
anyio==4.2.0
|
||||
appdirs==1.4.4
|
||||
argon2-cffi==23.1.0
|
||||
argon2-cffi-bindings==21.2.0
|
||||
arrow==1.3.0
|
||||
asttokens==2.4.1
|
||||
async-lru==2.0.4
|
||||
async-timeout==4.0.3
|
||||
attrs==23.2.0
|
||||
autodoc-pydantic==2.0.1
|
||||
azure-ai-formrecognizer==3.3.2
|
||||
azure-common==1.1.28
|
||||
azure-core==1.29.6
|
||||
azure-identity==1.17.1
|
||||
Babel==2.14.0
|
||||
beautifulsoup4==4.12.2
|
||||
black==23.12.1
|
||||
bleach==6.1.0
|
||||
blosc2==2.7.1
|
||||
build==1.0.3
|
||||
certifi==2023.11.17
|
||||
cffi==1.16.0
|
||||
charset-normalizer==3.3.2
|
||||
click==8.1.7
|
||||
colorama==0.4.6
|
||||
comm==0.2.2
|
||||
contourpy==1.2.1
|
||||
coverage==7.4.0
|
||||
cryptography==41.0.7
|
||||
cycler==0.12.1
|
||||
Cython==3.0.7
|
||||
dataclasses-json==0.6.3
|
||||
debugpy==1.8.2
|
||||
decorator==5.1.1
|
||||
defusedxml==0.7.1
|
||||
dill==0.3.8
|
||||
distro==1.9.0
|
||||
docker==7.1.0
|
||||
docutils==0.20.1
|
||||
exceptiongroup==1.2.0
|
||||
executing==2.0.1
|
||||
fastjsonschema==2.20.0
|
||||
feedparser==6.0.11
|
||||
filelock==3.13.1
|
||||
fire==0.5.0
|
||||
fonttools==4.53.1
|
||||
fqdn==1.5.1
|
||||
frozenlist==1.4.1
|
||||
fsspec==2023.12.2
|
||||
furo==2023.9.10
|
||||
fuzzywuzzy==0.18.0
|
||||
git-changelog==2.4.0
|
||||
greenlet==3.0.3
|
||||
h11==0.14.0
|
||||
httpcore==1.0.2
|
||||
httpx==0.26.0
|
||||
idna==3.6
|
||||
imagesize==1.4.1
|
||||
importlib-metadata==7.0.1
|
||||
iniconfig==2.0.0
|
||||
ipykernel==6.29.5
|
||||
ipython==8.26.0
|
||||
ipywidgets==8.1.3
|
||||
isodate==0.6.1
|
||||
isoduration==20.11.0
|
||||
isort==5.13.2
|
||||
jaraco.classes==3.3.0
|
||||
jedi==0.19.1
|
||||
jeepney==0.8.0
|
||||
joblib==1.4.2
|
||||
json5==0.9.25
|
||||
jsonpatch==1.33
|
||||
jsonpointer==2.4
|
||||
jsonschema==4.23.0
|
||||
jsonschema-specifications==2023.12.1
|
||||
jupyter==1.0.0
|
||||
jupyter-console==6.6.3
|
||||
jupyter-events==0.10.0
|
||||
jupyter-lsp==2.2.5
|
||||
jupyter_client==8.6.2
|
||||
jupyter_core==5.7.2
|
||||
jupyter_server==2.14.2
|
||||
jupyter_server_terminals==0.5.3
|
||||
jupyterlab==4.2.4
|
||||
jupyterlab_pygments==0.3.0
|
||||
jupyterlab_server==2.27.3
|
||||
jupyterlab_widgets==3.0.11
|
||||
keyring==24.3.0
|
||||
kiwisolver==1.4.5
|
||||
langchain==0.0.353
|
||||
langchain-community==0.0.7
|
||||
langchain-core==0.1.4
|
||||
langsmith==0.0.75
|
||||
Levenshtein==0.25.1
|
||||
livereload==2.6.3
|
||||
loguru==0.7.2
|
||||
loguru-mypy==0.0.4
|
||||
lxml==5.0.0
|
||||
markdown-it-py==3.0.0
|
||||
marshmallow==3.20.1
|
||||
matplotlib==3.9.1
|
||||
matplotlib-inline==0.1.7
|
||||
mdit-py-plugins==0.4.0
|
||||
mdurl==0.1.2
|
||||
mistune==3.0.2
|
||||
more-itertools==10.1.0
|
||||
msal==1.30.0
|
||||
msal-extensions==1.2.0
|
||||
msgpack==1.0.8
|
||||
msrest==0.7.1
|
||||
multidict==6.0.4
|
||||
mypy==1.10.0
|
||||
mypy-extensions==1.0.0
|
||||
myst-parser==2.0.0
|
||||
nbclient==0.10.0
|
||||
nbconvert==7.16.4
|
||||
nbformat==5.10.4
|
||||
ndindex==1.8
|
||||
nest-asyncio==1.6.0
|
||||
nh3==0.2.15
|
||||
notebook==7.2.1
|
||||
notebook_shim==0.2.4
|
||||
numexpr==2.10.1
|
||||
numpy==1.26.2
|
||||
oauthlib==3.2.2
|
||||
openai==1.6.1
|
||||
overrides==7.7.0
|
||||
packaging==23.2
|
||||
pandarallel==1.6.5
|
||||
pandas==2.1.4
|
||||
pandocfilters==1.5.1
|
||||
parso==0.8.4
|
||||
pathspec==0.12.1
|
||||
patsy==0.5.6
|
||||
pexpect==4.9.0
|
||||
pkginfo==1.9.6
|
||||
platformdirs==4.1.0
|
||||
pluggy==1.3.0
|
||||
portalocker==2.10.1
|
||||
prometheus_client==0.20.0
|
||||
prompt_toolkit==3.0.47
|
||||
psutil==6.0.0
|
||||
ptyprocess==0.7.0
|
||||
pure_eval==0.2.3
|
||||
py-cpuinfo==9.0.0
|
||||
pycparser==2.21
|
||||
pydantic==2.5.3
|
||||
pydantic-settings==2.1.0
|
||||
pydantic_core==2.14.6
|
||||
Pygments==2.17.2
|
||||
PyJWT==2.8.0
|
||||
PyMuPDF==1.24.9
|
||||
PyMuPDFb==1.24.9
|
||||
pyparsing==3.1.2
|
||||
pypdf==3.17.4
|
||||
pyproject_hooks==1.0.0
|
||||
pytest==7.4.4
|
||||
python-dateutil==2.8.2
|
||||
python-dotenv==1.0.0
|
||||
python-json-logger==2.0.7
|
||||
python-Levenshtein==0.25.1
|
||||
pytz==2023.3.post1
|
||||
PyYAML==6.0.1
|
||||
pyzmq==26.0.3
|
||||
qtconsole==5.5.2
|
||||
QtPy==2.4.1
|
||||
rapidfuzz==3.9.5
|
||||
readme-renderer==42.0
|
||||
referencing==0.35.1
|
||||
regex==2024.7.24
|
||||
requests==2.31.0
|
||||
requests-oauthlib==1.3.1
|
||||
requests-toolbelt==1.0.0
|
||||
rfc3339-validator==0.1.4
|
||||
rfc3986==2.0.0
|
||||
rfc3986-validator==0.1.1
|
||||
rich==13.7.0
|
||||
rpds-py==0.19.1
|
||||
ruamel.yaml==0.18.5
|
||||
ruamel.yaml.clib==0.2.8
|
||||
ruff==0.4.5
|
||||
scikit-learn==1.5.1
|
||||
SecretStorage==3.3.3
|
||||
semver==3.0.2
|
||||
Send2Trash==1.8.3
|
||||
setuptools-scm==8.0.4
|
||||
sgmllib3k==1.0.0
|
||||
shellingham==1.5.4
|
||||
six==1.16.0
|
||||
sniffio==1.3.0
|
||||
snowballstemmer==2.2.0
|
||||
soupsieve==2.5
|
||||
Sphinx==7.2.6
|
||||
sphinx-autobuild==2021.3.14
|
||||
sphinx-basic-ng==1.0.0b2
|
||||
sphinx-click==5.1.0
|
||||
sphinx-togglebutton==0.3.2
|
||||
sphinxcontrib-applehelp==1.0.7
|
||||
sphinxcontrib-devhelp==1.0.5
|
||||
sphinxcontrib-htmlhelp==2.0.4
|
||||
sphinxcontrib-jsmath==1.0.1
|
||||
sphinxcontrib-qthelp==1.0.6
|
||||
sphinxcontrib-serializinghtml==1.1.9
|
||||
SQLAlchemy==2.0.24
|
||||
stack-data==0.6.3
|
||||
statsmodels==0.14.2
|
||||
tables==3.9.2
|
||||
tabulate==0.9.0
|
||||
tenacity==8.2.3
|
||||
termcolor==2.4.0
|
||||
terminado==0.18.1
|
||||
threadpoolctl==3.5.0
|
||||
tiktoken==0.7.0
|
||||
tinycss2==1.3.0
|
||||
toml-sort==0.23.1
|
||||
tomli==2.0.1
|
||||
tomlkit==0.12.3
|
||||
tornado==6.4
|
||||
tqdm==4.66.1
|
||||
traitlets==5.14.3
|
||||
tree-sitter==0.22.3
|
||||
tree-sitter-python==0.21.0
|
||||
twine==4.0.2
|
||||
typer==0.9.0
|
||||
types-psutil==6.0.0.20240621
|
||||
types-python-dateutil==2.9.0.20240316
|
||||
types-PyYAML==6.0.12.20240724
|
||||
types-tqdm==4.66.0.20240417
|
||||
typing-inspect==0.9.0
|
||||
tzdata==2023.4
|
||||
uri-template==1.3.0
|
||||
urllib3==2.1.0
|
||||
wcwidth==0.2.13
|
||||
webcolors==24.6.0
|
||||
webencodings==0.5.1
|
||||
websocket-client==1.8.0
|
||||
widgetsnbextension==4.0.11
|
||||
yarl==1.9.4
|
||||
zipp==3.17.0
|
||||
dill==0.3.9
|
||||
pillow==10.4.0
|
||||
psutil==6.1.0
|
||||
rich==13.9.2
|
||||
scipy==1.14.1
|
||||
tqdm==4.66.5
|
||||
litellm==1.72.4
|
||||
|
||||
+7
-239
@@ -1,240 +1,8 @@
|
||||
aiohttp==3.9.1
|
||||
aiosignal==1.3.1
|
||||
alabaster==0.7.13
|
||||
annotated-types==0.6.0
|
||||
anyio==4.2.0
|
||||
appdirs==1.4.4
|
||||
argon2-cffi==23.1.0
|
||||
argon2-cffi-bindings==21.2.0
|
||||
arrow==1.3.0
|
||||
asttokens==2.4.1
|
||||
async-lru==2.0.4
|
||||
attrs==23.2.0
|
||||
autodoc-pydantic==2.0.1
|
||||
azure-ai-formrecognizer==3.3.2
|
||||
azure-common==1.1.28
|
||||
azure-core==1.29.6
|
||||
azure-identity==1.17.1
|
||||
Babel==2.14.0
|
||||
beautifulsoup4==4.12.2
|
||||
black==23.12.1
|
||||
bleach==6.1.0
|
||||
blosc2==2.7.1
|
||||
build==1.0.3
|
||||
certifi==2023.11.17
|
||||
cffi==1.16.0
|
||||
charset-normalizer==3.3.2
|
||||
click==8.1.7
|
||||
colorama==0.4.6
|
||||
comm==0.2.2
|
||||
contourpy==1.2.1
|
||||
coverage==7.4.0
|
||||
cryptography==41.0.7
|
||||
cycler==0.12.1
|
||||
Cython==3.0.7
|
||||
dataclasses-json==0.6.3
|
||||
debugpy==1.8.2
|
||||
decorator==5.1.1
|
||||
defusedxml==0.7.1
|
||||
dill==0.3.8
|
||||
distro==1.9.0
|
||||
docker==7.1.0
|
||||
docutils==0.20.1
|
||||
executing==2.0.1
|
||||
fastjsonschema==2.20.0
|
||||
feedparser==6.0.11
|
||||
filelock==3.13.1
|
||||
fire==0.5.0
|
||||
fonttools==4.53.1
|
||||
fqdn==1.5.1
|
||||
frozenlist==1.4.1
|
||||
fsspec==2023.12.2
|
||||
furo==2023.9.10
|
||||
fuzzywuzzy==0.18.0
|
||||
git-changelog==2.4.0
|
||||
greenlet==3.0.3
|
||||
h11==0.14.0
|
||||
httpcore==1.0.2
|
||||
httpx==0.26.0
|
||||
idna==3.6
|
||||
imagesize==1.4.1
|
||||
importlib-metadata==7.0.1
|
||||
iniconfig==2.0.0
|
||||
ipykernel==6.29.5
|
||||
ipython==8.26.0
|
||||
ipywidgets==8.1.3
|
||||
isodate==0.6.1
|
||||
isoduration==20.11.0
|
||||
isort==5.13.2
|
||||
jaraco.classes==3.3.0
|
||||
jedi==0.19.1
|
||||
jeepney==0.8.0
|
||||
joblib==1.4.2
|
||||
json5==0.9.25
|
||||
jsonpatch==1.33
|
||||
jsonpointer==2.4
|
||||
jsonschema==4.23.0
|
||||
jsonschema-specifications==2023.12.1
|
||||
jupyter==1.0.0
|
||||
jupyter-console==6.6.3
|
||||
jupyter-events==0.10.0
|
||||
jupyter-lsp==2.2.5
|
||||
jupyter_client==8.6.2
|
||||
jupyter_core==5.7.2
|
||||
jupyter_server==2.14.2
|
||||
jupyter_server_terminals==0.5.3
|
||||
jupyterlab==4.2.4
|
||||
jupyterlab_pygments==0.3.0
|
||||
jupyterlab_server==2.27.3
|
||||
jupyterlab_widgets==3.0.11
|
||||
keyring==24.3.0
|
||||
kiwisolver==1.4.5
|
||||
langchain==0.0.353
|
||||
langchain-community==0.0.7
|
||||
langchain-core==0.1.4
|
||||
langsmith==0.0.75
|
||||
Levenshtein==0.25.1
|
||||
livereload==2.6.3
|
||||
loguru==0.7.2
|
||||
loguru-mypy==0.0.4
|
||||
lxml==5.0.0
|
||||
markdown-it-py==3.0.0
|
||||
marshmallow==3.20.1
|
||||
matplotlib==3.9.1
|
||||
matplotlib-inline==0.1.7
|
||||
mdit-py-plugins==0.4.0
|
||||
mdurl==0.1.2
|
||||
mistune==3.0.2
|
||||
more-itertools==10.1.0
|
||||
msal==1.30.0
|
||||
msal-extensions==1.2.0
|
||||
msgpack==1.0.8
|
||||
msrest==0.7.1
|
||||
multidict==6.0.4
|
||||
mypy==1.10.0
|
||||
mypy-extensions==1.0.0
|
||||
myst-parser==2.0.0
|
||||
nbclient==0.10.0
|
||||
nbconvert==7.16.4
|
||||
nbformat==5.10.4
|
||||
ndindex==1.8
|
||||
nest-asyncio==1.6.0
|
||||
nh3==0.2.15
|
||||
notebook==7.2.1
|
||||
notebook_shim==0.2.4
|
||||
numexpr==2.10.1
|
||||
numpy==1.26.2
|
||||
oauthlib==3.2.2
|
||||
openai==1.6.1
|
||||
overrides==7.7.0
|
||||
packaging==23.2
|
||||
pandarallel==1.6.5
|
||||
pandas==2.1.4
|
||||
pandocfilters==1.5.1
|
||||
parso==0.8.4
|
||||
pathspec==0.12.1
|
||||
patsy==0.5.6
|
||||
pexpect==4.9.0
|
||||
pkginfo==1.9.6
|
||||
platformdirs==4.1.0
|
||||
pluggy==1.3.0
|
||||
portalocker==2.10.1
|
||||
prometheus_client==0.20.0
|
||||
prompt_toolkit==3.0.47
|
||||
psutil==6.0.0
|
||||
ptyprocess==0.7.0
|
||||
pure_eval==0.2.3
|
||||
py-cpuinfo==9.0.0
|
||||
pycparser==2.21
|
||||
pydantic==2.5.3
|
||||
pydantic-settings==2.1.0
|
||||
pydantic_core==2.14.6
|
||||
Pygments==2.17.2
|
||||
PyJWT==2.9.0
|
||||
PyMuPDF==1.24.9
|
||||
PyMuPDFb==1.24.9
|
||||
pyparsing==3.1.2
|
||||
pypdf==3.17.4
|
||||
pyproject_hooks==1.0.0
|
||||
pytest==7.4.4
|
||||
python-dateutil==2.8.2
|
||||
python-dotenv==1.0.0
|
||||
python-json-logger==2.0.7
|
||||
python-Levenshtein==0.25.1
|
||||
pytz==2023.3.post1
|
||||
PyYAML==6.0.1
|
||||
pyzmq==26.0.3
|
||||
qtconsole==5.5.2
|
||||
QtPy==2.4.1
|
||||
rapidfuzz==3.9.5
|
||||
readme-renderer==42.0
|
||||
referencing==0.35.1
|
||||
regex==2024.7.24
|
||||
requests==2.31.0
|
||||
requests-oauthlib==1.3.1
|
||||
requests-toolbelt==1.0.0
|
||||
rfc3339-validator==0.1.4
|
||||
rfc3986==2.0.0
|
||||
rfc3986-validator==0.1.1
|
||||
rich==13.7.0
|
||||
rpds-py==0.19.1
|
||||
ruamel.yaml==0.18.5
|
||||
ruamel.yaml.clib==0.2.8
|
||||
ruff==0.4.5
|
||||
scikit-learn==1.5.1
|
||||
SecretStorage==3.3.3
|
||||
semver==3.0.2
|
||||
Send2Trash==1.8.3
|
||||
setuptools-scm==8.0.4
|
||||
sgmllib3k==1.0.0
|
||||
shellingham==1.5.4
|
||||
six==1.16.0
|
||||
sniffio==1.3.0
|
||||
snowballstemmer==2.2.0
|
||||
soupsieve==2.5
|
||||
Sphinx==7.2.6
|
||||
sphinx-autobuild==2021.3.14
|
||||
sphinx-basic-ng==1.0.0b2
|
||||
sphinx-click==5.1.0
|
||||
sphinx-togglebutton==0.3.2
|
||||
sphinxcontrib-applehelp==1.0.7
|
||||
sphinxcontrib-devhelp==1.0.5
|
||||
sphinxcontrib-htmlhelp==2.0.4
|
||||
sphinxcontrib-jsmath==1.0.1
|
||||
sphinxcontrib-qthelp==1.0.6
|
||||
sphinxcontrib-serializinghtml==1.1.9
|
||||
SQLAlchemy==2.0.24
|
||||
stack-data==0.6.3
|
||||
statsmodels==0.14.2
|
||||
tables==3.9.2
|
||||
tabulate==0.9.0
|
||||
tenacity==8.2.3
|
||||
termcolor==2.4.0
|
||||
terminado==0.18.1
|
||||
threadpoolctl==3.5.0
|
||||
tiktoken==0.7.0
|
||||
tinycss2==1.3.0
|
||||
toml-sort==0.23.1
|
||||
tomlkit==0.12.3
|
||||
tornado==6.4
|
||||
tqdm==4.66.1
|
||||
traitlets==5.14.3
|
||||
tree-sitter==0.22.3
|
||||
tree-sitter-python==0.21.0
|
||||
twine==4.0.2
|
||||
typer==0.9.0
|
||||
types-psutil==6.0.0.20240621
|
||||
types-python-dateutil==2.9.0.20240316
|
||||
types-PyYAML==6.0.12.20240724
|
||||
types-tqdm==4.66.0.20240417
|
||||
typing-inspect==0.9.0
|
||||
tzdata==2023.4
|
||||
uri-template==1.3.0
|
||||
urllib3==2.1.0
|
||||
wcwidth==0.2.13
|
||||
webcolors==24.6.0
|
||||
webencodings==0.5.1
|
||||
websocket-client==1.8.0
|
||||
widgetsnbextension==4.0.11
|
||||
yarl==1.9.4
|
||||
zipp==3.17.0
|
||||
dill==0.3.9
|
||||
pillow==10.4.0
|
||||
psutil==6.1.0
|
||||
rich==13.9.2
|
||||
scipy==1.14.1
|
||||
tqdm==4.66.5
|
||||
litellm==1.72.4
|
||||
|
||||
Binary file not shown.
|
Before Width: | Height: | Size: 170 KiB |
Vendored
+332
@@ -0,0 +1,332 @@
|
||||
{
|
||||
"alpha053_15": {
|
||||
"description": "Reversal class factor, negative delta of a ratio involving close, low, and high prices over 15 days.",
|
||||
"formulation": "-1 times Deltaleft(frac{(text{close} - text{low}) - (text{high} - text{close})}{text{close} - text{low}}, 15right)",
|
||||
"variables": {
|
||||
"Delta(x, d)": "Change in 'x' over 'd' days.",
|
||||
"text{close}": "Closing price of the stock.",
|
||||
"text{low}": "Lowest price of the stock for the day.",
|
||||
"text{high}": "Highest price of the stock for the day."
|
||||
},
|
||||
"Category": "Volume&Price",
|
||||
"Difficulty": "Easy",
|
||||
"gt_code": "import pandas as pd\ndata_pv = pd.read_hdf('daily_pv.h5')\nnew_df= data_pv.reset_index()\n# Calculate Alpha053\nnew_df['ratio'] = (new_df['$close'] - new_df['$low'] - (new_df['$high'] - new_df['$close'])) / (new_df['$close'] - new_df['$low'])\n# the change of ratio in new_df over the 15 days\nnew_df['result']=-new_df['ratio'].diff(15)\n# transfer the result to series\nresult=pd.DataFrame(new_df['result']).set_index(data_pv.index)\nresult=result['result']\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"liquidity_imbalance": {
|
||||
"description": "liquidity_imbalance=std(minute trading liquidity_imbalance)/mean(minute trading liquidity_imbalance).",
|
||||
"formulation": "liquidity_imbalance = frac{text{std}(text{minute trading liquidity_imbalance})}{text{mean}(text{minute liquidity_imbalance})}",
|
||||
"variables": {
|
||||
"std(minute liquidity_imbalance)": "Standard deviation of trading liquidity_imbalance for each minute of the trading day.",
|
||||
"mean(minute liquidity_imbalance)": "Mean of trading liquidity_imbalance for each minute of the trading day.",
|
||||
"liquidity_imbalance": "(bid_size-ask_size)/(bid_size+ask_size), we use something like bidV for the size"
|
||||
},
|
||||
"Category": "High-Frequency",
|
||||
"Difficulty": "Medium",
|
||||
"gt_code": "import pandas as pd\ndata_hf = pd.read_hdf('high_freq.h5')\nsample_df= data_hf.reset_index()\n# Convert 'datetime' column to datetime and extract date for grouping\nsample_df['date'] = sample_df['datetime'].dt.date\nsample_df['liquidity_imbalance']=(sample_df['bidV']-sample_df['askV'])/(sample_df['bidV']+sample_df['askV'])\n# Group by instrument and date\ngrouped = sample_df.groupby(['date','instrument'])['liquidity_imbalance']\n# Calculate mean and standard deviation of the volume for each group\nstats = grouped.agg(['mean', 'std'])\n# Calculate Z value for each instrument per day\nstats['liquidity_imbalance'] = stats['std'] / stats['mean']\n# Display the calculated Z values\nresult=stats['liquidity_imbalance']\nresult.index.names = ['datetime','instrument']\n# result = result.swaplevel().sort_index()\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"liquidity_imbalance_2": {
|
||||
"description": "liquidity_imbalance=std(minute trading liquidity_imbalance)/mean(minute trading liquidity_imbalance).",
|
||||
"formulation": "liquidity_imbalance = frac{text{std}(text{minute trading liquidity_imbalance})}{text{mean}(text{minute liquidity_imbalance})}",
|
||||
"variables": {
|
||||
"std(minute liquidity_imbalance)": "Standard deviation of trading liquidity_imbalance for each minute of the trading day.",
|
||||
"mean(minute liquidity_imbalance)": "Mean of trading liquidity_imbalance for each minute of the trading day.",
|
||||
"liquidity_imbalance": "(bid_size-ask_size)/2*(bid_size+ask_size), we use something like bidV for the size"
|
||||
},
|
||||
"Category": "High-Frequency",
|
||||
"Difficulty": "Medium",
|
||||
"gt_code": "import pandas as pd\ndata_hf = pd.read_hdf('high_freq.h5')\nsample_df= data_hf.reset_index()\n# Convert 'datetime' column to datetime and extract date for grouping\nsample_df['date'] = sample_df['datetime'].dt.date\nsample_df['liquidity_imbalance']=(sample_df['bidV']-sample_df['askV'])/((sample_df['bidV']+sample_df['askV'])*2)\n# Group by instrument and date\ngrouped = sample_df.groupby(['date','instrument'])['liquidity_imbalance']\n# Calculate mean and standard deviation of the volume for each group\nstats = grouped.agg(['mean', 'std'])\n# Calculate Z value for each instrument per day\nstats['liquidity_imbalance'] = stats['std'] / stats['mean']\n# Display the calculated Z values\nresult=stats['liquidity_imbalance']\nresult.index.names = ['datetime','instrument']\n# result = result.swaplevel().sort_index()\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"liquidity_imbalance_3": {
|
||||
"description": "liquidity_imbalance=std(minute trading liquidity_imbalance)/mean(minute trading liquidity_imbalance).",
|
||||
"formulation": "liquidity_imbalance = frac{text{std}(text{minute trading liquidity_imbalance})}{text{mean}(text{minute liquidity_imbalance})}",
|
||||
"variables": {
|
||||
"std(minute liquidity_imbalance)": "Standard deviation of trading liquidity_imbalance for each minute of the trading day.",
|
||||
"mean(minute liquidity_imbalance)": "Mean of trading liquidity_imbalance for each minute of the trading day.",
|
||||
"liquidity_imbalance": "(bid_size-ask_size)/3*(bid_size+ask_size), we use something like bidV for the size"
|
||||
},
|
||||
"Category": "High-Frequency",
|
||||
"Difficulty": "Medium",
|
||||
"gt_code": "import pandas as pd\ndata_hf = pd.read_hdf('high_freq.h5')\nsample_df= data_hf.reset_index()\n# Convert 'datetime' column to datetime and extract date for grouping\nsample_df['date'] = sample_df['datetime'].dt.date\nsample_df['liquidity_imbalance']=(sample_df['bidV']-sample_df['askV'])/((sample_df['bidV']+sample_df['askV'])*3)\n# Group by instrument and date\ngrouped = sample_df.groupby(['date','instrument'])['liquidity_imbalance']\n# Calculate mean and standard deviation of the volume for each group\nstats = grouped.agg(['mean', 'std'])\n# Calculate Z value for each instrument per day\nstats['liquidity_imbalance'] = stats['std'] / stats['mean']\n# Display the calculated Z values\nresult=stats['liquidity_imbalance']\nresult.index.names = ['datetime','instrument']\n# result = result.swaplevel().sort_index()\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"micro_price": {
|
||||
"description": "micro_price=std(minute trading micro_price)/mean(minute trading micro_price).",
|
||||
"formulation": "micro_price = frac{text{std}(text{minute trading micro_price})}{text{mean}(text{minute micro_price})}",
|
||||
"variables": {
|
||||
"std(minute micro_price)": "Standard deviation of trading micro_price for each minute of the trading day.",
|
||||
"mean(minute micro_price)": "Mean of trading micro_price for each minute of the trading day.",
|
||||
"micro_price": "((df['bid_price'] * df['ask_size']) + (df['ask_price'] * df['bid_size'])) / (df['bid_size'] + df['ask_size'])"
|
||||
},
|
||||
"Category": "High-Frequency",
|
||||
"Difficulty": "Hard",
|
||||
"gt_code": "import pandas as pd\ndata_hf = pd.read_hdf('high_freq.h5')\nsample_df= data_hf.reset_index()\n# Convert 'datetime' column to datetime and extract date for grouping\nsample_df['date'] = sample_df['datetime'].dt.date\nsample_df['micro_price']=(sample_df['bid']*sample_df['askV']+sample_df['ask']*sample_df['bidV'])/(sample_df['bidV']+sample_df['askV'])\n# Group by instrument and date\ngrouped = sample_df.groupby(['date','instrument'])['micro_price']\n# Calculate mean and standard deviation of the volume for each group\nstats = grouped.agg(['mean', 'std'])\n# Calculate Z value for each instrument per day\nstats['micro_price'] = stats['std'] / stats['mean']\n# Display the calculated Z values\nresult=stats['micro_price']\nresult.index.names = ['datetime','instrument']\n# result = result.swaplevel().sort_index()\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"micro_price_2": {
|
||||
"description": "micro_price_2=std(minute trading micro_price)/mean(minute trading micro_price).",
|
||||
"formulation": "micro_price_2 = frac{text{std}(text{minute trading micro_price})}{text{mean}(text{minute micro_price})}",
|
||||
"variables": {
|
||||
"std(minute micro_price)": "Standard deviation of trading micro_price for each minute of the trading day.",
|
||||
"mean(minute micro_price)": "Mean of trading micro_price for each minute of the trading day.",
|
||||
"micro_price": "((df['bid_price'] * df['ask_size']) + (df['ask_price'] * df['bid_size'])) / 2*(df['bid_size'] + df['ask_size']), we use something like bidV for the size"
|
||||
},
|
||||
"Category": "High-Frequency",
|
||||
"Difficulty": "Hard",
|
||||
"gt_code": "import pandas as pd\ndata_hf = pd.read_hdf('high_freq.h5')\nsample_df= data_hf.reset_index()\n# Convert 'datetime' column to datetime and extract date for grouping\nsample_df['date'] = sample_df['datetime'].dt.date\nsample_df['micro_price']=(sample_df['bid']*sample_df['askV']+sample_df['ask']*sample_df['bidV'])/((sample_df['bidV']+sample_df['askV'])*2)\n# Group by instrument and date\ngrouped = sample_df.groupby(['date','instrument'])['micro_price']\n# Calculate mean and standard deviation of the volume for each group\nstats = grouped.agg(['mean', 'std'])\n# Calculate Z value for each instrument per day\nstats['micro_price'] = stats['std'] / stats['mean']\n# Display the calculated Z values\nresult=stats['micro_price']\nresult.index.names = ['datetime','instrument']\n# result = result.swaplevel().sort_index()\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"micro_price_3": {
|
||||
"description": "micro_price_3=std(minute trading micro_price)/mean(minute trading micro_price).",
|
||||
"formulation": "micro_price_3 = frac{text{std}(text{minute trading micro_price})}{text{mean}(text{minute micro_price})}",
|
||||
"variables": {
|
||||
"std(minute micro_price)": "Standard deviation of trading micro_price for each minute of the trading day.",
|
||||
"mean(minute micro_price)": "Mean of trading micro_price for each minute of the trading day.",
|
||||
"micro_price": "((df['bid_price'] * df['ask_size']) + (df['ask_price'] * df['bid_size'])) / 3*(df['bid_size'] + df['ask_size']), we use something like bidV for the size"
|
||||
},
|
||||
"Category": "High-Frequency",
|
||||
"Difficulty": "Hard",
|
||||
"gt_code": "import pandas as pd\ndata_hf = pd.read_hdf('high_freq.h5')\nsample_df= data_hf.reset_index()\n# Convert 'datetime' column to datetime and extract date for grouping\nsample_df['date'] = sample_df['datetime'].dt.date\nsample_df['micro_price']=(sample_df['bid']*sample_df['askV']+sample_df['ask']*sample_df['bidV'])/((sample_df['bidV']+sample_df['askV'])*3)\n# Group by instrument and date\ngrouped = sample_df.groupby(['date','instrument'])['micro_price']\n# Calculate mean and standard deviation of the volume for each group\nstats = grouped.agg(['mean', 'std'])\n# Calculate Z value for each instrument per day\nstats['micro_price'] = stats['std'] / stats['mean']\n# Display the calculated Z values\nresult=stats['micro_price']\nresult.index.names = ['datetime','instrument']\n# result = result.swaplevel().sort_index()\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"mid_price": {
|
||||
"description": "mid_price=std(minute trading mid_price)/mean(minute trading mid_price).",
|
||||
"formulation": "mid_price = frac{text{std}(text{minute trading mid price})}{text{mean}(text{minute mid price})}",
|
||||
"variables": {
|
||||
"std(minute mid_price)": "Standard deviation of trading mid_price for each minute of the trading day.",
|
||||
"mean(minute mid_price)": "Mean of trading mid_price for each minute of the trading day.",
|
||||
"mid_price": "The average of the bid and ask prices."
|
||||
},
|
||||
"Category": "High-Frequency",
|
||||
"Difficulty": "Easy",
|
||||
"gt_code": "import pandas as pd\ndata_hf = pd.read_hdf('high_freq.h5')\nsample_df= data_hf.reset_index()\n# Convert 'datetime' column to datetime and extract date for grouping\nsample_df['date'] = sample_df['datetime'].dt.date\nsample_df['mid_price']=(sample_df['bid']+sample_df['ask'])/2\n# Group by instrument and date\ngrouped = sample_df.groupby(['date','instrument'])['mid_price']\n# Calculate mean and standard deviation of the volume for each group\nstats = grouped.agg(['mean', 'std'])\nstats['mid_price'] = stats['std'] / stats['mean']\nresult=stats['mid_price']\nresult.index.names = ['datetime','instrument']\n# result = result.swaplevel().sort_index()\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"mid_price_2": {
|
||||
"description": "mid_price=std(minute trading mid_price)/mean(minute trading mid_price).",
|
||||
"formulation": "mid_price = frac{text{std}(text{minute trading mid price})}{text{mean}(text{minute mid price})}",
|
||||
"variables": {
|
||||
"std(minute mid_price)": "Standard deviation of trading mid_price for each minute of the trading day.",
|
||||
"mean(minute mid_price)": "Mean of trading mid_price for each minute of the trading day.",
|
||||
"mid_price_2": "the average of the bid and ask prices plus the the average of the bid and ask size (bidV and askV)."
|
||||
},
|
||||
"Category": "High-Frequency",
|
||||
"Difficulty": "Easy",
|
||||
"gt_code": "import pandas as pd\ndata_hf = pd.read_hdf('high_freq.h5')\nsample_df= data_hf.reset_index()\n# Convert 'datetime' column to datetime and extract date for grouping\nsample_df['date'] = sample_df['datetime'].dt.date\nsample_df['mid_price']=(sample_df['bid']+sample_df['ask'])/2+(sample_df['bidV']+sample_df['askV'])/2\n# Group by instrument and date\ngrouped = sample_df.groupby(['date','instrument'])['mid_price']\n# Calculate mean and standard deviation of the volume for each group\nstats = grouped.agg(['mean', 'std'])\nstats['mid_price'] = stats['std'] / stats['mean']\nresult=stats['mid_price']\nresult.index.names = ['datetime','instrument']\n# result = result.swaplevel().sort_index()\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"mid_price_3": {
|
||||
"description": "mid_price=std(minute trading mid_price)/mean(minute trading mid_price).",
|
||||
"formulation": "mid_price = frac{text{std}(text{minute trading mid price})}{text{mean}(text{minute mid price})}",
|
||||
"variables": {
|
||||
"std(minute mid_price)": "Standard deviation of trading mid_price for each minute of the trading day.",
|
||||
"mean(minute mid_price)": "Mean of trading mid_price for each minute of the trading day.",
|
||||
"mid_price_3": "The coefficient of variation (CV) of the mid-price for each minute of the trading day, calculated as the standard deviation of the mid-price divided by the mean mid-price."
|
||||
},
|
||||
"Category": "High-Frequency",
|
||||
"Difficulty": "Easy",
|
||||
"gt_code": "import pandas as pd\ndata_hf = pd.read_hdf('high_freq.h5')\nsample_df= data_hf.reset_index()\n# Convert 'datetime' column to datetime and extract date for grouping\nsample_df['date'] = sample_df['datetime'].dt.date\nsample_df['mid_price']=(sample_df['bid']+sample_df['ask'])/3\n# Group by instrument and date\ngrouped = sample_df.groupby(['date','instrument'])['mid_price']\n# Calculate mean and standard deviation of the volume for each group\nstats = grouped.agg(['mean', 'std'])\nstats['mid_price'] = stats['std'] / stats['mean']\nresult=stats['mid_price']\nresult.index.names = ['datetime','instrument']\n# result = result.swaplevel().sort_index()\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"PB_ROE": {
|
||||
"description": "Constructed using the ranking difference between PB and ROE, with regression versions of PB and ROE replacing original PB and ROE to obtain reconstructed factor values.",
|
||||
"formulation": "text{rank}(PB_t) - rank(ROE_t)",
|
||||
"variables": {
|
||||
"text{rank}(PB_t)": "Ranking of regression version PB on cross-section at time t.",
|
||||
"text{rank}(ROE_t)": "Ranking of regression version single-quarter ROE on cross-section at time t."
|
||||
},
|
||||
"Category": "Fundamentals",
|
||||
"Difficulty": "Easy",
|
||||
"gt_code": "import pandas as pd\ndata_f = pd.read_hdf('daily_f.h5')\ndata = data_f.reset_index()\n# Calculate the rank of PB and ROE\ndata['PB_rank'] = data.groupby('datetime')['B/P'].rank()\ndata['ROE_rank'] = data.groupby('datetime')['ROE'].rank()\n# Calculate the difference between the ranks\ndata['PB_ROE'] = data['PB_rank'] - data['ROE_rank']\n# set the datetime and instrument as index and drop the original index\nresult=pd.DataFrame(data['PB_ROE']).set_index(data_f.index)\n# transfer the result to series\nresult=result['PB_ROE']\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"PB_ROE_2": {
|
||||
"description": "Constructed using the ranking difference between PB/2 and ROE, with regression versions of PB and ROE replacing original PB and ROE to obtain reconstructed factor values.",
|
||||
"formulation": "text{rank}(PB_t)/2 - rank(ROE_t)",
|
||||
"variables": {
|
||||
"text{rank}(PB_t)": "Ranking of regression version PB on cross-section at time t.",
|
||||
"text{rank}(ROE_t)": "Ranking of regression version single-quarter ROE on cross-section at time t."
|
||||
},
|
||||
"Category": "Fundamentals",
|
||||
"Difficulty": "Easy",
|
||||
"gt_code": "import pandas as pd\ndata_f = pd.read_hdf('daily_f.h5')\ndata = data_f.reset_index()\n# Calculate the rank of PB and ROE\ndata['PB_rank'] = data.groupby('datetime')['B/P'].rank()\ndata['ROE_rank'] = data.groupby('datetime')['ROE'].rank()\n# Calculate the difference between the ranks\ndata['PB_ROE'] = data['PB_rank']/2 - data['ROE_rank']\n# set the datetime and instrument as index and drop the original index\nresult=pd.DataFrame(data['PB_ROE']).set_index(data_f.index)\n# transfer the result to series\nresult=result['PB_ROE']\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"PB_ROE_3": {
|
||||
"description": "Constructed using the ranking difference between PB/3 and ROE, with regression versions of PB and ROE replacing original PB and ROE to obtain reconstructed factor values.",
|
||||
"formulation": "text{rank}(PB_t)/3 - rank(ROE_t)",
|
||||
"variables": {
|
||||
"text{rank}(PB_t)": "Ranking of regression version PB on cross-section at time t.",
|
||||
"text{rank}(ROE_t)": "Ranking of regression version single-quarter ROE on cross-section at time t."
|
||||
},
|
||||
"Category": "Fundamentals",
|
||||
"Difficulty": "Easy",
|
||||
"gt_code": "import pandas as pd\ndata_f = pd.read_hdf('daily_f.h5')\ndata = data_f.reset_index()\n# Calculate the rank of PB and ROE\ndata['PB_rank'] = data.groupby('datetime')['B/P'].rank()\ndata['ROE_rank'] = data.groupby('datetime')['ROE'].rank()\n# Calculate the difference between the ranks\ndata['PB_ROE'] = data['PB_rank']/3 - data['ROE_rank']\n# set the datetime and instrument as index and drop the original index\nresult=pd.DataFrame(data['PB_ROE']).set_index(data_f.index)\n# transfer the result to series\nresult=result['PB_ROE']\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"PB_ROE_movement": {
|
||||
"description": "PB_ROE_movement=five day PB_ROE movement indicator(-1 and 1 or 0).",
|
||||
"formulation": "PB_ROE_movement = 5_day_movement(PB_ROE), PB_ROE = text{rank}(PB_t) - rank(ROE_t)",
|
||||
"variables": {
|
||||
"PB_ROE": "the ranking difference between PB and ROE.",
|
||||
"5_day_PB_ROE_movement": "1 if PB_ROE is higher than the PB_ROE 5 days ago, -1 if PB_ROE is lower than the PB_ROE 5 days ago, 0 if PB_ROE is the same as the PB_ROE 5 days ago.",
|
||||
"text{rank}(PB_t)": "Ranking of regression version PB on cross-section at time t.",
|
||||
"text{rank}(ROE_t)": "Ranking of regression version single-quarter ROE on cross-section at time t."
|
||||
},
|
||||
"Category": "Fundamentals",
|
||||
"Difficulty": "Hard",
|
||||
"gt_code": "import pandas as pd\ndata_f = pd.read_hdf('daily_f.h5')\nsample_df = data_f.reset_index()\n# Calculate the rank of PB and ROE\nsample_df['PB_rank'] = sample_df.groupby('datetime')['B/P'].rank()\nsample_df['ROE_rank'] = sample_df.groupby('datetime')['ROE'].rank()\nsample_df['PB_ROE'] = sample_df['PB_rank'] - sample_df['ROE_rank']\n# Group by instrument and date\nsample_df['PB_ROE_movement'] = sample_df['PB_ROE'].diff(periods=5).apply(lambda x: 1 if x > 0 else (-1 if x < 0 else 0))\n#calculate the mid_price_movement ratio for each day\n# set the datetime and instrument as index and drop the original index\nresult=pd.DataFrame(sample_df['PB_ROE_movement']).set_index(data_f.index)\n# transfer the result to series\nresult=result['PB_ROE_movement']\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"PB_ROE_movement_10": {
|
||||
"description": "PB_ROE_movement=10 days PB_ROE movement indicator(-1 and 1 or 0).",
|
||||
"formulation": "PB_ROE_movement = 10_day_movement(PB_ROE), PB_ROE = text{rank}(PB_t) - rank(ROE_t)",
|
||||
"variables": {
|
||||
"PB_ROE": "the ranking difference between PB and ROE.",
|
||||
"10_day_PB_ROE_movement": "1 if PB_ROE is higher than the PB_ROE 10 days ago, -1 if PB_ROE is lower than the PB_ROE 10 days ago, 0 if PB_ROE is the same as the PB_ROE 10 days ago.",
|
||||
"text{rank}(PB_t)": "Ranking of regression version PB on cross-section at time t.",
|
||||
"text{rank}(ROE_t)": "Ranking of regression version single-quarter ROE on cross-section at time t."
|
||||
},
|
||||
"Category": "Fundamentals",
|
||||
"Difficulty": "Hard",
|
||||
"gt_code": "import pandas as pd\ndata_f = pd.read_hdf('daily_f.h5')\nsample_df = data_f.reset_index()\n# Calculate the rank of PB and ROE\nsample_df['PB_rank'] = sample_df.groupby('datetime')['B/P'].rank()\nsample_df['ROE_rank'] = sample_df.groupby('datetime')['ROE'].rank()\nsample_df['PB_ROE'] = sample_df['PB_rank'] - sample_df['ROE_rank']\n# Group by instrument and date\nsample_df['PB_ROE_movement'] = sample_df['PB_ROE'].diff(periods=10).apply(lambda x: 1 if x > 0 else (-1 if x < 0 else 0))\n#calculate the mid_price_movement ratio for each day\n# set the datetime and instrument as index and drop the original index\nresult=pd.DataFrame(sample_df['PB_ROE_movement']).set_index(data_f.index)\n# transfer the result to series\nresult=result['PB_ROE_movement']\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"PB_ROE_movement_20": {
|
||||
"description": "PB_ROE_movement=20 days PB_ROE movement indicator(-1 and 1 or 0).",
|
||||
"formulation": "PB_ROE_movement = 20_day_movement(PB_ROE), PB_ROE = text{rank}(PB_t) - rank(ROE_t)",
|
||||
"variables": {
|
||||
"PB_ROE": "the ranking difference between PB and ROE.",
|
||||
"20_day_PB_ROE_movement": "1 if PB_ROE is higher than the PB_ROE 20 days ago, -1 if PB_ROE is lower than the PB_ROE 20 days ago, 0 if PB_ROE is the same as the PB_ROE 20 days ago.",
|
||||
"text{rank}(PB_t)": "Ranking of regression version PB on cross-section at time t.",
|
||||
"text{rank}(ROE_t)": "Ranking of regression version single-quarter ROE on cross-section at time t."
|
||||
},
|
||||
"Category": "Fundamentals",
|
||||
"Difficulty": "Hard",
|
||||
"gt_code": "import pandas as pd\ndata_f = pd.read_hdf('daily_f.h5')\nsample_df = data_f.reset_index()\n# Calculate the rank of PB and ROE\nsample_df['PB_rank'] = sample_df.groupby('datetime')['B/P'].rank()\nsample_df['ROE_rank'] = sample_df.groupby('datetime')['ROE'].rank()\nsample_df['PB_ROE'] = sample_df['PB_rank'] - sample_df['ROE_rank']\n# Group by instrument and date\nsample_df['PB_ROE_movement'] = sample_df['PB_ROE'].diff(periods=20).apply(lambda x: 1 if x > 0 else (-1 if x < 0 else 0))\n#calculate the mid_price_movement ratio for each day\n# set the datetime and instrument as index and drop the original index\nresult=pd.DataFrame(sample_df['PB_ROE_movement']).set_index(data_f.index)\n# transfer the result to series\nresult=result['PB_ROE_movement']\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"ROE_movement": {
|
||||
"description": "ROE_movement=five day ROE movement indicator(-1 and 1 or 0).",
|
||||
"formulation": "ROE_movement = 5_day_movement(ROE)",
|
||||
"variables": {
|
||||
"ROE": "ROE in fundamental statistics.",
|
||||
"5_day_ROE_movement": "1 if ROE is higher than the ROE 5 days ago, -1 if ROE is lower than the ROE 5 days ago, 0 if ROE is the same as the ROE 5 days ago."
|
||||
},
|
||||
"Category": "Fundamentals",
|
||||
"Difficulty": "Medium",
|
||||
"gt_code": "import pandas as pd\ndata_f = pd.read_hdf('daily_f.h5')\nsample_df = data_f.reset_index()\n# Group by instrument and date\nsample_df['ROE_movement'] = sample_df['ROE'].diff(periods=5).apply(lambda x: 1 if x > 0 else (-1 if x < 0 else 0))\n#calculate the mid_price_movement ratio for each day\n# set the datetime and instrument as index and drop the original index\nresult=pd.DataFrame(sample_df['ROE_movement']).set_index(data_f.index)\n# transfer the result to series\nresult=result['ROE_movement']\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"ROE_movement_10": {
|
||||
"description": "ROE_movement_10=ten day ROE movement indicator(-1 and 1 or 0).",
|
||||
"formulation": "ROE_movement = 10_day_movement(ROE)",
|
||||
"variables": {
|
||||
"ROE": "ROE in fundamental statistics.",
|
||||
"10_day_ROE_movement": "1 if ROE is higher than the ROE 10 days ago, -1 if ROE is lower than the ROE 10 days ago, 0 if ROE is the same as the ROE 10 days ago."
|
||||
},
|
||||
"Category": "Fundamentals",
|
||||
"Difficulty": "Medium",
|
||||
"gt_code": "import pandas as pd\ndata_f = pd.read_hdf('daily_f.h5')\nsample_df = data_f.reset_index()\n# Group by instrument and date\nsample_df['ROE_movement'] = sample_df['ROE'].diff(periods=10).apply(lambda x: 1 if x > 0 else (-1 if x < 0 else 0))\n#calculate the mid_price_movement ratio for each day\n# set the datetime and instrument as index and drop the original index\nresult=pd.DataFrame(sample_df['ROE_movement']).set_index(data_f.index)\n# transfer the result to series\nresult=result['ROE_movement']\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"ROE_movement_20": {
|
||||
"description": "ROE_movement_20=20 day ROE movement indicator(-1 and 1 or 0).",
|
||||
"formulation": "ROE_movement_20 = 20_day_movement(ROE)",
|
||||
"variables": {
|
||||
"ROE": "ROE in fundamental statistics.",
|
||||
"20_day_ROE_movement": "1 if ROE is higher than the ROE 20 days ago, -1 if ROE is lower than the ROE 20 days ago, 0 if ROE is the same as the ROE 20 days ago."
|
||||
},
|
||||
"Category": "Fundamentals",
|
||||
"Difficulty": "Medium",
|
||||
"gt_code": "import pandas as pd\ndata_f = pd.read_hdf('daily_f.h5')\nsample_df = data_f.reset_index()\n# Group by instrument and date\nsample_df['ROE_movement'] = sample_df['ROE'].diff(periods=20).apply(lambda x: 1 if x > 0 else (-1 if x < 0 else 0))\n#calculate the mid_price_movement ratio for each day\n# set the datetime and instrument as index and drop the original index\nresult=pd.DataFrame(sample_df['ROE_movement']).set_index(data_f.index)\n# transfer the result to series\nresult=result['ROE_movement']\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"alpha_pv_diff": {
|
||||
"description": "alpha_pv_diff is defined as the ratio of the difference between close prices 10 days change and open prices 10 days change to the sum of the highest minus lowest prices plus a small constant.",
|
||||
"formulation": "frac{(text{close_diff10} - text{open_diff10})}{(text{high} - text{low} + 0.001)}",
|
||||
"variables": {
|
||||
"close": "Closing price of the stock",
|
||||
"open": "Opening price of the stock",
|
||||
"high": "Highest price of the stock during the day",
|
||||
"low": "Lowest price of the stock during the day"
|
||||
},
|
||||
"Category": "Volume&Price",
|
||||
"Difficulty": "Medium",
|
||||
"gt_code": "import pandas as pd\ndata_pv = pd.read_hdf('daily_pv.h5')\nnew_df= data_pv.reset_index()\n# Calculate Alpha101\nnew_df['result'] = (new_df['$close'].diff(10) - new_df['$open'].diff(10)) / (new_df['$high'] - new_df['$low'] + 0.001)\n# keep the index of the original dataframe\nresult=pd.DataFrame(new_df['result']).set_index(data_pv.index)\n# transfer the result to series\nresult=result['result']\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"alpha_pv_diff_15": {
|
||||
"description": "alpha_pv_diff is defined as the ratio of the difference between close prices 15 days change and open prices 15 days change to the sum of the highest minus lowest prices plus a small constant.",
|
||||
"formulation": "frac{(text{close_diff15} - text{open_diff15})}{(text{high} - text{low} + 0.001)}",
|
||||
"variables": {
|
||||
"close": "Closing price of the stock",
|
||||
"open": "Opening price of the stock",
|
||||
"high": "Highest price of the stock during the day",
|
||||
"low": "Lowest price of the stock during the day"
|
||||
},
|
||||
"Category": "Volume&Price",
|
||||
"Difficulty": "Medium",
|
||||
"gt_code": "import pandas as pd\ndata_pv = pd.read_hdf('daily_pv.h5')\nnew_df= data_pv.reset_index()\n# Calculate Alpha101\nnew_df['result'] = (new_df['$close'].diff(15) - new_df['$open'].diff(15)) / (new_df['$high'] - new_df['$low'] + 0.001)\n# keep the index of the original dataframe\nresult=pd.DataFrame(new_df['result']).set_index(data_pv.index)\n# transfer the result to series\nresult=result['result']\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"alpha_pv_diff_20": {
|
||||
"description": "alpha_pv_diff is defined as the ratio of the difference between close prices 20 days change and open prices 20 days change to the sum of the highest minus lowest prices plus a small constant.",
|
||||
"formulation": "frac{(text{close_diff20} - text{open_diff20})}{(text{high} - text{low} + 0.001)}",
|
||||
"variables": {
|
||||
"close": "Closing price of the stock",
|
||||
"open": "Opening price of the stock",
|
||||
"high": "Highest price of the stock during the day",
|
||||
"low": "Lowest price of the stock during the day"
|
||||
},
|
||||
"Category": "Volume&Price",
|
||||
"Difficulty": "Medium",
|
||||
"gt_code": "import pandas as pd\ndata_pv = pd.read_hdf('daily_pv.h5')\nnew_df= data_pv.reset_index()\n# Calculate Alpha101\nnew_df['result'] = (new_df['$close'].diff(20) - new_df['$open'].diff(20)) / (new_df['$high'] - new_df['$low'] + 0.001)\n# keep the index of the original dataframe\nresult=pd.DataFrame(new_df['result']).set_index(data_pv.index)\n# transfer the result to series\nresult=result['result']\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"alpha_pv_diff_pct": {
|
||||
"description": "alpha_pv is defined as the ratio of the difference between close prices 10 days change and open prices 10 days change to the sum of the highest prices 10 days change ratio minus lowest prices 10 days change ratio plus a small constant.",
|
||||
"formulation": "frac{(text{close_diff10} - text{open_diff10})}{(text{high_pct10} - text{low_pct10} + 0.001)}",
|
||||
"variables": {
|
||||
"close": "Closing price of the stock",
|
||||
"open": "Opening price of the stock",
|
||||
"high": "Highest price of the stock during the day",
|
||||
"low": "Lowest price of the stock during the day"
|
||||
},
|
||||
"Category": "Volume&Price",
|
||||
"Difficulty": "Hard",
|
||||
"gt_code": "import pandas as pd\ndata_pv = pd.read_hdf('daily_pv.h5')\nnew_df= data_pv.reset_index()\n# Calculate Alpha101\nnew_df['result'] = (new_df['$close'].diff(10) - new_df['$open'].diff(10)) / (new_df['$high'].pct_change(10) - new_df['$low'].pct_change(10) + 0.001)\n# keep the index of the original dataframe\nresult=pd.DataFrame(new_df['result']).set_index(data_pv.index)\n# transfer the result to series\nresult=result['result']\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"alpha_pv_diff_pct_15": {
|
||||
"description": "alpha_pv is defined as the ratio of the difference between close prices 15 days change and open prices 15 days change to the sum of the highest prices 10 days change ratio minus lowest prices 10 days change ratio plus a small constant.",
|
||||
"formulation": "frac{(text{close_diff15} - text{open_diff15})}{(text{high_pct10} - text{low_pct10} + 0.001)}",
|
||||
"variables": {
|
||||
"close": "Closing price of the stock",
|
||||
"open": "Opening price of the stock",
|
||||
"high": "Highest price of the stock during the day",
|
||||
"low": "Lowest price of the stock during the day"
|
||||
},
|
||||
"Category": "Volume&Price",
|
||||
"Difficulty": "Hard",
|
||||
"gt_code": "import pandas as pd\ndata_pv = pd.read_hdf('daily_pv.h5')\nnew_df= data_pv.reset_index()\n# Calculate Alpha101\nnew_df['result'] = (new_df['$close'].diff(15) - new_df['$open'].diff(15)) / (new_df['$high'].pct_change(10) - new_df['$low'].pct_change(10) + 0.001)\n# keep the index of the original dataframe\nresult=pd.DataFrame(new_df['result']).set_index(data_pv.index)\n# transfer the result to series\nresult=result['result']\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"alpha_pv_diff_pct_20": {
|
||||
"description": "alpha_pv is defined as the ratio of the difference between close prices 20 days change and open prices 20 days change to the sum of the highest prices 10 days change ratio minus lowest prices 10 days change ratio plus a small constant.",
|
||||
"formulation": "frac{(text{close_diff20} - text{open_diff20})}{(text{high_pct10} - text{low_pct10} + 0.001)}",
|
||||
"variables": {
|
||||
"close": "Closing price of the stock",
|
||||
"open": "Opening price of the stock",
|
||||
"high": "Highest price of the stock during the day",
|
||||
"low": "Lowest price of the stock during the day"
|
||||
},
|
||||
"Category": "Volume&Price",
|
||||
"Difficulty": "Hard",
|
||||
"gt_code": "import pandas as pd\ndata_pv = pd.read_hdf('daily_pv.h5')\nnew_df= data_pv.reset_index()\n# Calculate Alpha101\nnew_df['result'] = (new_df['$close'].diff(20) - new_df['$open'].diff(20)) / (new_df['$high'].pct_change(10) - new_df['$low'].pct_change(10) + 0.001)\n# keep the index of the original dataframe\nresult=pd.DataFrame(new_df['result']).set_index(data_pv.index)\n# transfer the result to series\nresult=result['result']\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"alpha053": {
|
||||
"description": "Reversal class factor, negative delta of a ratio involving close, low, and high prices over 9 days.",
|
||||
"formulation": "-1 times Deltaleft(frac{(text{close} - text{low}) - (text{high} - text{close})}{text{close} - text{low}}, 9right)",
|
||||
"variables": {
|
||||
"Delta(x, d)": "Change in 'x' over 'd' days.",
|
||||
"text{close}": "Closing price of the stock.",
|
||||
"text{low}": "Lowest price of the stock for the day.",
|
||||
"text{high}": "Highest price of the stock for the day."
|
||||
},
|
||||
"Category": "Volume&Price",
|
||||
"Difficulty": "Easy",
|
||||
"gt_code": "import pandas as pd\ndata_pv = pd.read_hdf('daily_pv.h5')\nnew_df= data_pv.reset_index()\n# Calculate Alpha053\nnew_df['ratio'] = (new_df['$close'] - new_df['$low'] - (new_df['$high'] - new_df['$close'])) / (new_df['$close'] - new_df['$low'])\n# the change of ratio in new_df over the 9 days\nnew_df['result']=-new_df['ratio'].diff(9)\n# transfer the result to series\nresult=pd.DataFrame(new_df['result']).set_index(data_pv.index)\nresult=result['result']\nresult.to_hdf('result.h5', key='data')"
|
||||
},
|
||||
"alpha053_5": {
|
||||
"description": "Reversal class factor, negative delta of a ratio involving close, low, and high prices over 5 days.",
|
||||
"formulation": "-1 times Deltaleft(frac{(text{close} - text{low}) - (text{high} - text{close})}{text{close} - text{low}}, 5right)",
|
||||
"variables": {
|
||||
"Delta(x, d)": "Change in 'x' over 'd' days.",
|
||||
"text{close}": "Closing price of the stock.",
|
||||
"text{low}": "Lowest price of the stock for the day.",
|
||||
"text{high}": "Highest price of the stock for the day."
|
||||
},
|
||||
"Category": "Volume&Price",
|
||||
"Difficulty": "Easy",
|
||||
"gt_code": "import pandas as pd\ndata_pv = pd.read_hdf('daily_pv.h5')\nnew_df= data_pv.reset_index()\n# Calculate Alpha053\nnew_df['ratio'] = (new_df['$close'] - new_df['$low'] - (new_df['$high'] - new_df['$close'])) / (new_df['$close'] - new_df['$low'])\n# the change of ratio in new_df over the 5 days\nnew_df['result']=-new_df['ratio'].diff(5)\n# transfer the result to series\nresult=pd.DataFrame(new_df['result']).set_index(data_pv.index)\nresult=result['result']\nresult.to_hdf('result.h5', key='data')"
|
||||
}
|
||||
}
|
||||
Vendored
BIN
Binary file not shown.
|
Before Width: | Height: | Size: 94 KiB After Width: | Height: | Size: 88 KiB |
@@ -11,13 +11,89 @@ Installation
|
||||
- for dev users: `See development <development.html>`_
|
||||
|
||||
**Install Docker**: RDAgent is designed for research and development, acting like a human researcher and developer. It can write and run code in various environments, primarily using Docker for code execution. This keeps the remaining dependencies simple. Users must ensure Docker is installed before attempting most scenarios. Please refer to the `official 🐳Docker page <https://docs.docker.com/engine/install/>`_ for installation instructions.
|
||||
Ensure the current user can run Docker commands **without using sudo**. You can verify this by executing `docker run hello-world`.
|
||||
|
||||
Configuration
|
||||
=============
|
||||
LiteLLM Backend Configuration (Default)
|
||||
=======================================
|
||||
|
||||
Option 1: Unified API base for both models
|
||||
------------------------------------------
|
||||
|
||||
.. code-block:: Properties
|
||||
|
||||
# Set to any model supported by LiteLLM.
|
||||
CHAT_MODEL=gpt-4o
|
||||
EMBEDDING_MODEL=text-embedding-3-small
|
||||
# Configure unified API base
|
||||
# The backend api_key fully follows the convention of litellm.
|
||||
OPENAI_API_BASE=<your_unified_api_base>
|
||||
OPENAI_API_KEY=<replace_with_your_openai_api_key>
|
||||
|
||||
Option 2: Separate API bases for Chat and Embedding models
|
||||
----------------------------------------------------------
|
||||
|
||||
.. code-block:: Properties
|
||||
|
||||
# Set to any model supported by LiteLLM.
|
||||
|
||||
# CHAT MODEL:
|
||||
CHAT_MODEL=gpt-4o
|
||||
OPENAI_API_BASE=<your_chat_api_base>
|
||||
OPENAI_API_KEY=<replace_with_your_openai_api_key>
|
||||
|
||||
# EMBEDDING MODEL:
|
||||
# TAKE siliconflow as an example, you can use other providers.
|
||||
# Note: embedding requires litellm_proxy prefix
|
||||
EMBEDDING_MODEL=litellm_proxy/BAAI/bge-large-en-v1.5
|
||||
LITELLM_PROXY_API_KEY=<replace_with_your_siliconflow_api_key>
|
||||
LITELLM_PROXY_API_BASE=https://api.siliconflow.cn/v1
|
||||
|
||||
Necessary parameters include:
|
||||
|
||||
- `CHAT_MODEL`: The model name of the chat model.
|
||||
|
||||
- `EMBEDDING_MODEL`: The model name of the embedding model.
|
||||
|
||||
- `OPENAI_API_BASE`: The base URL of the API. If `EMBEDDING_MODEL` does not start with `litellm_proxy/`, this is used for both chat and embedding models; otherwise, it is used for `CHAT_MODEL` only.
|
||||
|
||||
Optional parameters (required if your embedding model is provided by a different provider than `CHAT_MODEL`):
|
||||
|
||||
- `LITELLM_PROXY_API_KEY`: The API key for the embedding model, required if `EMBEDDING_MODEL` starts with `litellm_proxy/`.
|
||||
|
||||
- `LITELLM_PROXY_API_BASE`: The base URL for the embedding model, required if `EMBEDDING_MODEL` starts with `litellm_proxy/`.
|
||||
|
||||
**Note:** If you are using an embedding model from a provider different from the chat model, remember to add the `litellm_proxy/` prefix to the `EMBEDDING_MODEL` name.
|
||||
|
||||
|
||||
The `CHAT_MODEL` and `EMBEDDING_MODEL` parameters will be passed into LiteLLM's completion function.
|
||||
|
||||
Therefore, when utilizing models provided by different providers, first review the interface configuration of LiteLLM. The model names must match those allowed by LiteLLM.
|
||||
|
||||
Additionally, you need to set up the the additional parameters for the respective model provider, and the parameter names must align with those required by LiteLLM.
|
||||
|
||||
For example, if you are using a DeepSeek model, you need to set as follows:
|
||||
|
||||
.. code-block:: Properties
|
||||
|
||||
# For some models LiteLLM requires a prefix to the model name.
|
||||
CHAT_MODEL=deepseek/deepseek-chat
|
||||
DEEPSEEK_API_KEY=<replace_with_your_deepseek_api_key>
|
||||
|
||||
Besides, when you are using reasoning models, the response might include the thought process. For this case, you need to set the following environment variable:
|
||||
|
||||
.. code-block:: Properties
|
||||
|
||||
REASONING_THINK_RM=True
|
||||
|
||||
For more details on LiteLLM requirements, refer to the `official LiteLLM documentation <https://docs.litellm.ai/docs>`_.
|
||||
|
||||
|
||||
Configuration(deprecated)
|
||||
=========================
|
||||
|
||||
To run the application, please create a `.env` file in the root directory of the project and add environment variables according to your requirements.
|
||||
|
||||
The standard configuration options for the user using the OpenAI API are provided in the `.env.example` file.
|
||||
If you are using this deprecated version, you should set `BACKEND` to `rdagent.oai.backend.DeprecBackend`.
|
||||
|
||||
Here are some other configuration options that you can use:
|
||||
|
||||
@@ -38,22 +114,23 @@ Azure OpenAI
|
||||
The following environment variables are standard configuration options for the user using the OpenAI API.
|
||||
|
||||
.. code-block:: Properties
|
||||
|
||||
|
||||
USE_AZURE=True
|
||||
|
||||
OPENAI_API_KEY=<replace_with_your_openai_api_key>
|
||||
|
||||
EMBEDDING_OPENAI_API_KEY=<replace_with_your_azure_openai_api_key>
|
||||
EMBEDDING_AZURE_API_BASE= # The endpoint for the Azure OpenAI API.
|
||||
EMBEDDING_AZURE_API_VERSION= # The version of the Azure OpenAI API.
|
||||
EMBEDDING_MODEL=text-embedding-3-small
|
||||
EMBEDDING_AZURE_API_BASE= # The base URL for the Azure OpenAI API.
|
||||
EMBEDDING_AZURE_API_VERSION = # The version of the Azure OpenAI API.
|
||||
|
||||
CHAT_MODEL=gpt-4-turbo
|
||||
CHAT_AZURE_API_VERSION = # The version of the Azure OpenAI API.
|
||||
CHAT_OPENAI_API_KEY=<replace_with_your_azure_openai_api_key>
|
||||
CHAT_AZURE_API_BASE= # The endpoint for the Azure OpenAI API.
|
||||
CHAT_AZURE_API_VERSION= # The version of the Azure OpenAI API.
|
||||
CHAT_MODEL= # The model name of the Azure OpenAI API.
|
||||
|
||||
Use Azure Token Provider
|
||||
------------------------
|
||||
|
||||
If you are using the Azure token provider, you need to set the `USE_AZURE_TOKEN_PROVIDER` environment variable to `True`. then
|
||||
If you are using the Azure token provider, you need to set the `CHAT_USE_AZURE_TOKEN_PROVIDER` and `EMBEDDING_USE_AZURE_TOKEN_PROVIDER` environment variable to `True`. then
|
||||
use the environment variables provided in the `Azure Configuration section <installation_and_configuration.html#azure-openai>`_.
|
||||
|
||||
|
||||
@@ -80,31 +157,33 @@ Configuration List
|
||||
|
||||
- OpenAI API Setting
|
||||
|
||||
+-----------------------------+--------------------------------------------------+-------------------------+
|
||||
| Configuration Option | Meaning | Default Value |
|
||||
+=============================+==================================================+=========================+
|
||||
| OPENAI_API_KEY | API key for both chat and embedding models | None |
|
||||
+-----------------------------+--------------------------------------------------+-------------------------+
|
||||
| EMBEDDING_OPENAI_API_KEY | Use a different API key for embedding model | None |
|
||||
+-----------------------------+--------------------------------------------------+-------------------------+
|
||||
| CHAT_OPENAI_API_KEY | Set to use a different API key for chat model | None |
|
||||
+-----------------------------+--------------------------------------------------+-------------------------+
|
||||
| EMBEDDING_MODEL | Name of the embedding model | text-embedding-3-small |
|
||||
+-----------------------------+--------------------------------------------------+-------------------------+
|
||||
| CHAT_MODEL | Name of the chat model | gpt-4-turbo |
|
||||
+-----------------------------+--------------------------------------------------+-------------------------+
|
||||
| EMBEDDING_AZURE_API_BASE | Base URL for the Azure OpenAI API | None |
|
||||
+-----------------------------+--------------------------------------------------+-------------------------+
|
||||
| EMBEDDING_AZURE_API_VERSION | Version of the Azure OpenAI API | None |
|
||||
+-----------------------------+--------------------------------------------------+-------------------------+
|
||||
| CHAT_AZURE_API_BASE | Base URL for the Azure OpenAI API | None |
|
||||
+-----------------------------+--------------------------------------------------+-------------------------+
|
||||
| CHAT_AZURE_API_VERSION | Version of the Azure OpenAI API | None |
|
||||
+-----------------------------+--------------------------------------------------+-------------------------+
|
||||
| USE_AZURE | True if you are using Azure OpenAI | False |
|
||||
+-----------------------------+--------------------------------------------------+-------------------------+
|
||||
| USE_AZURE_TOKEN_PROVIDER | True if you are using a Azure Token Provider | False |
|
||||
+-----------------------------+--------------------------------------------------+-------------------------+
|
||||
+-----------------------------------+-----------------------------------------------------------------+-------------------------+
|
||||
| Configuration Option | Meaning | Default Value |
|
||||
+===================================+=================================================================+=========================+
|
||||
| OPENAI_API_KEY | API key for both chat and embedding models | None |
|
||||
+-----------------------------------+-----------------------------------------------------------------+-------------------------+
|
||||
| EMBEDDING_OPENAI_API_KEY | Use a different API key for embedding model | None |
|
||||
+-----------------------------------+-----------------------------------------------------------------+-------------------------+
|
||||
| CHAT_OPENAI_API_KEY | Set to use a different API key for chat model | None |
|
||||
+-----------------------------------+-----------------------------------------------------------------+-------------------------+
|
||||
| EMBEDDING_MODEL | Name of the embedding model | text-embedding-3-small |
|
||||
+-----------------------------------+-----------------------------------------------------------------+-------------------------+
|
||||
| CHAT_MODEL | Name of the chat model | gpt-4-turbo |
|
||||
+-----------------------------------+-----------------------------------------------------------------+-------------------------+
|
||||
| EMBEDDING_AZURE_API_BASE | Base URL for the Azure OpenAI API | None |
|
||||
+-----------------------------------+-----------------------------------------------------------------+-------------------------+
|
||||
| EMBEDDING_AZURE_API_VERSION | Version of the Azure OpenAI API | None |
|
||||
+-----------------------------------+-----------------------------------------------------------------+-------------------------+
|
||||
| CHAT_AZURE_API_BASE | Base URL for the Azure OpenAI API | None |
|
||||
+-----------------------------------+-----------------------------------------------------------------+-------------------------+
|
||||
| CHAT_AZURE_API_VERSION | Version of the Azure OpenAI API | None |
|
||||
+-----------------------------------+-----------------------------------------------------------------+-------------------------+
|
||||
| USE_AZURE | True if you are using Azure OpenAI | False |
|
||||
+-----------------------------------+-----------------------------------------------------------------+-------------------------+
|
||||
| CHAT_USE_AZURE_TOKEN_PROVIDER | True if you are using an Azure Token Provider in chat model | False |
|
||||
+-----------------------------------+-----------------------------------------------------------------+-------------------------+
|
||||
| EMBEDDING_USE_AZURE_TOKEN_PROVIDER| True if you are using an Azure Token Provider in embedding model| False |
|
||||
+-----------------------------------+-----------------------------------------------------------------+-------------------------+
|
||||
|
||||
- Globol Setting
|
||||
|
||||
|
||||
+21
-30
@@ -5,21 +5,12 @@ Benchmark
|
||||
Introduction
|
||||
=============
|
||||
|
||||
|
||||
Benchmarking the capabilities of the R&D is a very important research problem of the research area.
|
||||
|
||||
Currently we are continuously exploring how to benchmark them.
|
||||
|
||||
The current benchmarks are listed in this page
|
||||
|
||||
Benchmarking the capabilities of R&D is a crucial research problem in this area. We are continuously exploring methods to benchmark these capabilities. The current benchmarks are listed on this page.
|
||||
|
||||
Development Capability Benchmarking
|
||||
===================================
|
||||
|
||||
|
||||
Benchmark is used to evaluate the effectiveness of factors with fixed data.
|
||||
|
||||
It mainly includes the following steps:
|
||||
Benchmarking is used to evaluate the effectiveness of factors with fixed data. It mainly includes the following steps:
|
||||
|
||||
1. :ref:`read and prepare the eval_data <data>`
|
||||
|
||||
@@ -27,34 +18,31 @@ It mainly includes the following steps:
|
||||
|
||||
3. :ref:`declare the eval method and pass the arguments <config>`
|
||||
|
||||
4. :ref:`run the eval <run>`
|
||||
4. :ref:`run the eval <run>`
|
||||
|
||||
5. :ref:`save and show the result <show>`
|
||||
5. :ref:`save and show the result <show>`
|
||||
|
||||
Configuration
|
||||
Configuration
|
||||
-------------
|
||||
.. _config:
|
||||
|
||||
.. autopydantic_settings:: rdagent.components.benchmark.conf.BenchmarkSettings
|
||||
|
||||
Example
|
||||
++++++++
|
||||
+++++++
|
||||
.. _example:
|
||||
|
||||
The default value for ``bench_test_round`` is 10, and it will take about 2 hours to run 10 rounds.
|
||||
To modify it from ``10`` to ``2`` you can adjust this by adding environment variables in the .env file as shown below.
|
||||
The default value for ``bench_test_round`` is 10, which takes about 2 hours to run. To modify it from ``10`` to ``2``, adjust the environment variables in the .env file as shown below.
|
||||
|
||||
.. code-block:: Properties
|
||||
|
||||
BENCHMARK_BENCH_TEST_ROUND=1
|
||||
BENCHMARK_BENCH_TEST_ROUND=2
|
||||
|
||||
Data Format
|
||||
-------------
|
||||
.. _data:
|
||||
|
||||
The sample data in ``bench_data_path`` is a dictionary where each key represents a factor name.
|
||||
|
||||
The value associated with each key is factor data containing the following information:
|
||||
The sample data in ``bench_data_path`` is a dictionary where each key represents a factor name. The value associated with each key is factor data containing the following information:
|
||||
|
||||
- **description**: A textual description of the factor.
|
||||
- **formulation**: A LaTeX formula representing the model's formulation.
|
||||
@@ -63,22 +51,24 @@ The value associated with each key is factor data containing the following infor
|
||||
- **Difficulty**: The difficulty level of implementing or understanding the factor.
|
||||
- **gt_code**: A piece of code associated with the factor.
|
||||
|
||||
Here is the example of this data format:
|
||||
Here is an example of this data format:
|
||||
|
||||
.. literalinclude:: ../../rdagent/components/benchmark/example.json
|
||||
:language: json
|
||||
|
||||
Ensure the data is placed in the ``FACTOR_COSTEER_SETTINGS.data_folder_debug``. The data files should be in ``.h5`` or ``.md`` format and must not be stored in any subfolders. LLM-Agents will review the file content and implement the tasks.
|
||||
|
||||
.. TODO: Add a script to automatically generate the data in the `rdagent/app/quant_factor_benchmark/data` folder.
|
||||
|
||||
Run Benchmark
|
||||
-------------
|
||||
.. _run:
|
||||
|
||||
Start benchmark after finishing the :doc:`../installation_and_configuration`.
|
||||
Start the benchmark after completing the :doc:`../installation_and_configuration`.
|
||||
|
||||
.. code-block:: Properties
|
||||
|
||||
python rdagent/app/quant_factor_benchmark/eval.py
|
||||
|
||||
|
||||
dotenv run -- python rdagent/app/benchmark/factor/eval.py
|
||||
|
||||
Once completed, a pkl file will be generated, and its path will be printed on the last line of the console.
|
||||
|
||||
@@ -86,18 +76,16 @@ Show Result
|
||||
-------------
|
||||
.. _show:
|
||||
|
||||
The ``analysis.py`` script is used to read data from pkl and convert it to an image.
|
||||
Modify the python code in ``rdagent/app/quant_factor_benchmark/analysis.py`` to specify the path to the pkl file and the output path for the png file.
|
||||
The ``analysis.py`` script reads data from the pkl file and converts it to an image. Modify the Python code in ``rdagent/app/quant_factor_benchmark/analysis.py`` to specify the path to the pkl file and the output path for the png file.
|
||||
|
||||
.. code-block:: Properties
|
||||
|
||||
python rdagent/app/quant_factor_benchmark/analysis.py
|
||||
dotenv run -- python rdagent/app/benchmark/factor/analysis.py <log/path to.pkl>
|
||||
|
||||
A png file will be saved to the designated path as shown below.
|
||||
|
||||
.. image:: ../_static/benchmark.png
|
||||
|
||||
|
||||
Related Paper
|
||||
-------------
|
||||
|
||||
@@ -116,3 +104,6 @@ Related Paper
|
||||
}
|
||||
|
||||
.. image:: https://github.com/user-attachments/assets/494f55d3-de9e-4e73-ba3d-a787e8f9e841
|
||||
|
||||
To replicate the benchmark detailed in the paper, please consult the factors listed in the following file: `RD2bench.json <../_static/RD2bench.json>`_.
|
||||
Please note use ``only_correct_format=False`` when evaluating the results.
|
||||
|
||||
+18
-19
@@ -13,34 +13,33 @@ In the two key areas of data-driven scenarios, model implementation and data bui
|
||||
The supported scenarios are listed below:
|
||||
|
||||
|
||||
|
||||
.. list-table::
|
||||
.. list-table::
|
||||
:header-rows: 1
|
||||
|
||||
* - Scenario/Target
|
||||
- Model Implementation
|
||||
- Data Building
|
||||
* - 💹 Finance
|
||||
- :ref:`🤖Iteratively Proposing Ideas & Evolving <model_agent_fin>`
|
||||
- :ref:`🦾Auto reports reading & implementation <data_copilot_fin>`
|
||||
|
||||
- :ref:`🥇The First Data-Centric Quant Multi-Agent Framework <quant_agent_fin>`
|
||||
- :ref:`🤖Iteratively Proposing Ideas & Evolving <model_agent_fin>`
|
||||
|
||||
:ref:`🦾Auto reports reading & implementation <data_copilot_fin>`
|
||||
|
||||
:ref:`🤖Iteratively Proposing Ideas & Evolving <data_agent_fin>`
|
||||
* - 🩺 Medical
|
||||
- :ref:`🤖Iteratively Proposing Ideas & Evolving <model_agent_med>`
|
||||
-
|
||||
* - 🏭 General
|
||||
- :ref:`🦾Auto paper reading & implementation <model_copilot_general>`
|
||||
-
|
||||
- :ref:`🦾Auto paper reading & implementation <model_copilot_general>`
|
||||
|
||||
- :ref:`🤖 Data Science <data_science_agent>`
|
||||
|
||||
|
||||
.. toctree::
|
||||
:maxdepth: 1
|
||||
:caption: Doctree:
|
||||
:hidden:
|
||||
|
||||
data_agent_fin
|
||||
data_copilot_fin
|
||||
model_agent_fin
|
||||
model_agent_med
|
||||
model_copilot_general
|
||||
:maxdepth: 1
|
||||
:caption: Doctree:
|
||||
:hidden:
|
||||
|
||||
quant_agent_fin
|
||||
data_agent_fin
|
||||
data_copilot_fin
|
||||
model_agent_fin
|
||||
model_copilot_general
|
||||
data_science
|
||||
|
||||
@@ -131,8 +131,8 @@ The following environment variables can be set in the `.env` file to customize t
|
||||
:settings-show-field-summary: False
|
||||
:exclude-members: Config
|
||||
|
||||
.. autopydantic_settings:: rdagent.components.coder.factor_coder.config.FactorImplementSettings
|
||||
.. autopydantic_settings:: rdagent.components.coder.factor_coder.config.FactorCoSTEERSettings
|
||||
:settings-show-field-summary: False
|
||||
:members: coder_use_cache, data_folder, data_folder_debug, file_based_execution_timeout, select_method, select_threshold, max_loop, knowledge_base_path, new_knowledge_base_path
|
||||
:members: coder_use_cache, data_folder, data_folder_debug, file_based_execution_timeout, select_method, max_loop, knowledge_base_path, new_knowledge_base_path
|
||||
:exclude-members: Config, fail_task_trial_limit, v1_query_former_trace_limit, v1_query_similar_success_limit, v2_query_component_limit, v2_query_error_limit, v2_query_former_trace_limit, v2_error_summary, v2_knowledge_sampler
|
||||
:no-index:
|
||||
|
||||
@@ -157,8 +157,8 @@ The following environment variables can be set in the `.env` file to customize t
|
||||
:show-inheritance:
|
||||
:exclude-members: Config
|
||||
|
||||
.. autopydantic_settings:: rdagent.components.coder.factor_coder.config.FactorImplementSettings
|
||||
.. autopydantic_settings:: rdagent.components.coder.factor_coder.config.FactorCoSTEERSettings
|
||||
:settings-show-field-summary: False
|
||||
:members: coder_use_cache, data_folder, data_folder_debug, file_based_execution_timeout, select_method, select_threshold, max_loop, knowledge_base_path, new_knowledge_base_path
|
||||
:members: coder_use_cache, data_folder, data_folder_debug, file_based_execution_timeout, select_method, max_loop, knowledge_base_path, new_knowledge_base_path
|
||||
:exclude-members: Config, python_bin, fail_task_trial_limit, v1_query_former_trace_limit, v1_query_similar_success_limit, v2_query_component_limit, v2_query_error_limit, v2_query_former_trace_limit, v2_error_summary, v2_knowledge_sampler
|
||||
:no-index:
|
||||
|
||||
@@ -0,0 +1,158 @@
|
||||
.. _data_science_agent:
|
||||
|
||||
=======================
|
||||
Data Science Agent
|
||||
=======================
|
||||
|
||||
**🤖 Automated Feature Engineering & Model Tuning Evolution**
|
||||
------------------------------------------------------------------------------------------
|
||||
The Data Science Agent is an agent that can automatically perform feature engineering and model tuning. It can be used to solve various data science problems, such as image classification, time series forecasting, and text classification.
|
||||
|
||||
🧭 Example Guide
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
- 🔧 **Set up RD-Agent Environment**
|
||||
|
||||
- Before you start, please make sure you have installed RD-Agent and configured the environment for RD-Agent correctly. If you want to know how to install and configure the RD-Agent, please refer to the `documentation <../installation_and_configuration.html>`_.
|
||||
|
||||
- 🔩 **Setting the Environment variables at .env file**
|
||||
|
||||
- Determine the path where the data will be stored and add it to the ``.env`` file.
|
||||
|
||||
.. code-block:: sh
|
||||
|
||||
dotenv set DS_LOCAL_DATA_PATH <your local directory>/ds_data
|
||||
dotenv set DS_SCEN rdagent.scenarios.data_science.scen.DataScienceScen
|
||||
|
||||
- 📥 **Prepare Competition Data**
|
||||
|
||||
- Data Science competition data typically consists of three components: a competition description file (in Markdown format), the competition dataset, and evaluation scripts. For reference, an example of a custom user-defined dataset is provided in ``rdagent/scenarios/data_science/example``.
|
||||
|
||||
- **Correct directory structure (Here is an example of competition data with id custom_data)**
|
||||
|
||||
.. code-block:: text
|
||||
|
||||
ds_data
|
||||
└── eval
|
||||
| └── custom_data
|
||||
| └── grade.py
|
||||
| └── valid.py
|
||||
| └── test.csv
|
||||
└── custom_data
|
||||
└── train.csv
|
||||
└── test.csv
|
||||
└── sample_submission.csv
|
||||
└── description.md
|
||||
└── sample.py
|
||||
|
||||
- ``ds_data/custom_data/train.csv:`` Necessary training data in csv or parquet format, or training images.
|
||||
|
||||
- ``ds_data/custom_data/description.md:`` (Optional) Competition description file.
|
||||
|
||||
- ``ds_data/custom_data/sample_submission.csv:`` (Optional) Competition sample submission file.
|
||||
|
||||
- ``ds_data/custom_data/sample.py:`` (Optional) Sample code for generating debug data from the competition dataset. If not provided, R&D-Agent will use its default sampling logic. For details, see the ``create_debug_data`` function in ``rdagent/scenarios/data_science/debug/data.py``.
|
||||
|
||||
- ``ds_data/eval/custom_data/grade.py:`` (Optional) Competition grade script, in order to calculate the score for the submission.
|
||||
|
||||
- ``ds_data/eval/custom_data/valid.py:`` (Optional) Competition validation script, in order to check if the submission format is correct.
|
||||
|
||||
- ``ds_data/eval/custom_data/submission_test.csv:`` (Optional) Competition test label file.
|
||||
|
||||
- 🔧 **Set up Environment for Custom User-defined Dataset**
|
||||
|
||||
.. code-block:: sh
|
||||
|
||||
dotenv set DS_SCEN rdagent.scenarios.data_science.scen.DataScienceScen
|
||||
dotenv set DS_LOCAL_DATA_PATH rdagent/scenarios/data_science/example
|
||||
dotenv set DS_IF_USING_MLE_DATA False
|
||||
dotenv set DS_CODER_ON_WHOLE_PIPELINE True
|
||||
dotenv set DS_CODER_COSTEER_ENV_TYPE docker
|
||||
|
||||
🔍 MLE-bench Guide: Running ML Engineering via MLE-bench
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
- 📝 **MLE-bench Overview**
|
||||
|
||||
- MLE-bench is a comprehensive benchmark designed to evaluate the ML engineering capabilities of AI systems using real-world scenarios. The dataset comprises 75 Kaggle competitions. Since Kaggle does not provide held-out test sets for these competitions, the benchmark includes preparation scripts that split the publicly available training data into new training and test sets, and grading scripts are provided for each competition to accurately evaluate submission scores.
|
||||
|
||||
- 🔧 **Set up Environment for MLE-bench**
|
||||
|
||||
- Running R&D-Agent on MLE-bench is designed for full automation. There is no need for manual downloads and data preparation. Simply set the environment variable ``DS_IF_USING_MLE_DATA`` to True.
|
||||
|
||||
- At runtime, R&D-Agent will automatically build the Docker image specified at ``rdagent/scenarios/kaggle/docker/mle_bench_docker/Dockerfile``. This image is responsible for downloading the required datasets and grading files for MLE-bench.
|
||||
|
||||
- Note: The first run may take longer than subsequent runs as the Docker image and data are being downloaded and set up for the first time.
|
||||
|
||||
.. code-block:: sh
|
||||
|
||||
dotenv set DS_LOCAL_DATA_PATH <your local directory>/ds_data
|
||||
dotenv set DS_IF_USING_MLE_DATA True
|
||||
|
||||
- 🔨 **Configuring the Kaggle API**
|
||||
|
||||
- Downloading Kaggle competition data requires the Kaggle API. You can set up the Kaggle API by following these steps:
|
||||
|
||||
- Register and login on the `Kaggle <https://www.kaggle.com/>`_ website.
|
||||
|
||||
- Click on the avatar (usually in the top right corner of the page) -> ``Settings`` -> ``Create New Token``, A file called ``kaggle.json`` will be downloaded.
|
||||
|
||||
- Move ``kaggle.json`` to ``~/.config/kaggle/``
|
||||
|
||||
- Modify the permissions of the ``kaggle.json`` file.
|
||||
|
||||
.. code-block:: sh
|
||||
|
||||
chmod 600 ~/.config/kaggle/kaggle.json
|
||||
|
||||
- For more information about Kaggle API Settings, refer to the `Kaggle API <https://github.com/Kaggle/kaggle-api>`_.
|
||||
|
||||
|
||||
- 🔩 **Setting the Environment Variables for MLE-bench**
|
||||
|
||||
- In addition to auto-downloading the benchmark data, you must also configure the runtime environment for executing the competition code.
|
||||
- Use the environment variable ``DS_CODER_COSTEER_ENV_TYPE`` to select the execution mode:
|
||||
|
||||
• When set to docker (the default), RD-Agent utilizes the official Kaggle Docker image (``gcr.io/kaggle-gpu-images/python:latest``) to ensure that all required packages are available.
|
||||
• If you prefer to use a custom Docker setup, you can modify the configuration using ``DS_DOCKER_IMAGE`` or ``DS_DOCKERFILE_FOLDER_PATH``.
|
||||
• Alternatively, if your competition work only demands basic libraries, you may set ``DS_CODER_COSTEER_ENV_TYPE`` to conda. In this mode, you must create a local conda environment named “kaggle” and pre-install the necessary packages. RD-Agent will execute the competition code within this “kaggle” conda environment.
|
||||
|
||||
.. code-block:: sh
|
||||
|
||||
# Configure the runtime environment: choice between 'docker' (default) or 'conda'
|
||||
dotenv set DS_CODER_COSTEER_ENV_TYPE docker
|
||||
|
||||
- 🚀 **Run the Application**
|
||||
|
||||
- You can directly run the application by using the following command:
|
||||
|
||||
.. code-block:: sh
|
||||
|
||||
rdagent data_science --competition <Competition ID>
|
||||
|
||||
- 📥 **Visualize the R&D Process**
|
||||
|
||||
- We provide a web UI to visualize the log. You just need to run:
|
||||
|
||||
.. code-block:: sh
|
||||
|
||||
streamlit run rdagent/log/ui/dsapp.py
|
||||
|
||||
- Then you can input the log path and visualize the R&D process.
|
||||
|
||||
- **Additional Guidance**
|
||||
|
||||
- **Combine different LLM Models at R&D Stage**
|
||||
|
||||
- You can combine different LLM models at the R&D stage.
|
||||
|
||||
- By default, when you set environment variable ``CHAT_MODEL``, it covers both R&D stages. When customizing the model for the development stage, you can set:
|
||||
|
||||
.. code-block:: sh
|
||||
|
||||
# This example sets the model to "o3-mini". For some models, the reasoning effort shoule be set to "None".
|
||||
dotenv set LITELLM_CHAT_MODEL_MAP '{"coding":{"model":"o3-mini","reasoning_effort":"high"},"running":{"model":"o3-mini","reasoning_effort":"high"}}'
|
||||
|
||||
|
||||
|
||||
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 152 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 12 KiB |
@@ -1,128 +0,0 @@
|
||||
.. _model_agent_med:
|
||||
|
||||
=======================
|
||||
Medical Model Agent
|
||||
=======================
|
||||
|
||||
**🤖 Automated Medical Predtion Model Evolution**
|
||||
------------------------------------------------------------------------------------------
|
||||
|
||||
📖 Background
|
||||
~~~~~~~~~~~~~~
|
||||
In this scenario, we consider the problem of risk prediction from patients' ICU monitoring data. We use the a public EHR dataset - MIMIC-III and extract a binary classification task for evaluating the framework.
|
||||
In this task, we aim at predicting the whether the patients will suffer from Acute Respiratory Failure (ARF) based their first 12 hours ICU monitoring data.
|
||||
|
||||
🎥 `Demo <https://rdagent.azurewebsites.net/dmm>`_
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. raw:: html
|
||||
|
||||
<div style="display: flex; justify-content: center; align-items: center;">
|
||||
<video width="600" controls>
|
||||
<source src="https://rdagent.azurewebsites.net/media/1653542fc1b9fa14a306c35c1b1fc48288f980793f38abe82b023af9.mp4" type="video/mp4">
|
||||
Your browser does not support the video tag.
|
||||
</video>
|
||||
</div>
|
||||
|
||||
|
||||
🌟 Introduction
|
||||
~~~~~~~~~~~~~~~~
|
||||
|
||||
In this scenario, our automated system proposes hypothesis, constructs model, implements code, receives back-testing, and uses feedbacks.
|
||||
Hypothesis is iterated in this continuous process.
|
||||
The system aims to automatically optimise performance metrics of medical prediction thereby finding the optimised code through autonomous research and development.
|
||||
|
||||
Here's an enhanced outline of the steps:
|
||||
|
||||
**Step 1 : Hypothesis Generation 🔍**
|
||||
|
||||
- Generate and propose initial hypotheses based on previous experiment analysis and domain expertise, with thorough reasoning and justification.
|
||||
|
||||
**Step 2 : Model Creation ✨**
|
||||
|
||||
- Transform the hypothesis into a model.
|
||||
- Develop, define, and implement a machine learning model, including its name, description, and formulation.
|
||||
|
||||
**Step 3 : Model Implementation 👨💻**
|
||||
|
||||
- Implement the model code based on the detailed description.
|
||||
- Evolve the model iteratively as a developer would, ensuring accuracy and efficiency.
|
||||
|
||||
**Step 4 : Backtesting with MIMIC-III 📉**
|
||||
|
||||
- Conduct backtesting using the newly developed model on the extracted task from MIMIC-III.
|
||||
- Evaluate the model's effectiveness and performance in terms of AUROC score.
|
||||
|
||||
**Step 5 : Feedback Analysis 🔍**
|
||||
|
||||
- Analyze backtest results to assess performance.
|
||||
- Incorporate feedback to refine hypotheses and improve the model.
|
||||
|
||||
**Step 6 :Hypothesis Refinement ♻️**
|
||||
|
||||
- Refine hypotheses based on feedback from backtesting.
|
||||
- Repeat the process to continuously improve the model.
|
||||
|
||||
⚡ Quick Start
|
||||
~~~~~~~~~~~~~~~~~
|
||||
|
||||
Please refer to the installation part in :doc:`../installation_and_configuration` to prepare your system dependency.
|
||||
|
||||
You can try our demo by running the following command:
|
||||
|
||||
- 🐍 Create a Conda Environment
|
||||
|
||||
- Create a new conda environment with Python (3.10 and 3.11 are well tested in our CI):
|
||||
|
||||
.. code-block:: sh
|
||||
|
||||
conda create -n rdagent python=3.10
|
||||
|
||||
- Activate the environment:
|
||||
|
||||
.. code-block:: sh
|
||||
|
||||
conda activate rdagent
|
||||
|
||||
- 📦 Install the RDAgent
|
||||
|
||||
- You can install the RDAgent package from PyPI:
|
||||
|
||||
.. code-block:: sh
|
||||
|
||||
pip install rdagent
|
||||
|
||||
- 📦 Request PhysioNet Account
|
||||
|
||||
- Apply for an account at `PhysioNet <https://physionet.org/>`_.
|
||||
- Request access to FIDDLE preprocessed data: `FIDDLE Dataset <https://physionet.org/content/mimic-eicu-fiddle-feature/1.0.0/>`_.
|
||||
- Place your username and password in `.env`.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
cat << EOF >> .env
|
||||
DM_USERNAME=<your_username>
|
||||
DM_PASSWORD=<your_password>
|
||||
EOF
|
||||
|
||||
|
||||
- 🚀 Run the Application
|
||||
|
||||
- You can directly run the application by using the following command:
|
||||
|
||||
.. code-block:: sh
|
||||
|
||||
rdagent med_model
|
||||
|
||||
🛠️ Usage of modules
|
||||
~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. _Env Config:
|
||||
|
||||
- **Env Config**
|
||||
|
||||
The following environment variables can be set in the `.env` file to customize the application's behavior:
|
||||
|
||||
.. autopydantic_settings:: rdagent.app.data_mining.conf.MedBasePropSetting
|
||||
:settings-show-field-summary: False
|
||||
:exclude-members: Config
|
||||
@@ -0,0 +1,113 @@
|
||||
.. _quant_agent_fin:
|
||||
|
||||
=====================
|
||||
Finance Quant Agent
|
||||
=====================
|
||||
|
||||
|
||||
**🥇The First Data-Centric Quant Multi-Agent Framework RD-Agent(Q)**
|
||||
---------------------------------------------------------------------
|
||||
|
||||
R&D-Agent for Quantitative Finance, in short **RD-Agent(Q)**, is the first data-centric, multi-agent framework designed to automate the full-stack research and development of quantitative strategies via coordinated factor-model co-optimization.
|
||||
|
||||
You can learn more details about **RD-Agent(Q)** through the `paper <https://arxiv.org/abs/2505.15155>`_.
|
||||
|
||||
⚡ Quick Start
|
||||
~~~~~~~~~~~~~~~~~
|
||||
|
||||
Before you start, please make sure you have installed RD-Agent and configured the environment for RD-Agent correctly. If you want to know how to install and configure the RD-Agent, please refer to the `documentation <../installation_and_configuration.html>`_.
|
||||
|
||||
Then, you can run the framework by running the following command:
|
||||
|
||||
- 🐍 Create a Conda Environment
|
||||
|
||||
- Create a new conda environment with Python (3.10 and 3.11 are well tested in our CI):
|
||||
|
||||
.. code-block:: sh
|
||||
|
||||
conda create -n rdagent python=3.10
|
||||
|
||||
- Activate the environment:
|
||||
|
||||
.. code-block:: sh
|
||||
|
||||
conda activate rdagent
|
||||
|
||||
- 📦 Install the RDAgent
|
||||
|
||||
- You can install the RDAgent package from PyPI:
|
||||
|
||||
.. code-block:: sh
|
||||
|
||||
pip install rdagent
|
||||
|
||||
- 🚀 Run the Application
|
||||
|
||||
- You can directly run the application by using the following command:
|
||||
|
||||
.. code-block:: sh
|
||||
|
||||
rdagent fin_quant
|
||||
|
||||
|
||||
🛠️ Usage of modules
|
||||
~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. _Env Config:
|
||||
|
||||
- **Env Config**
|
||||
|
||||
The following environment variables can be set in the `.env` file to customize the application's behavior:
|
||||
|
||||
.. autopydantic_settings:: rdagent.app.qlib_rd_loop.conf.QuantBasePropSetting
|
||||
:settings-show-field-summary: False
|
||||
:exclude-members: Config
|
||||
|
||||
.. autopydantic_settings:: rdagent.components.coder.factor_coder.config.FactorCoSTEERSettings
|
||||
:settings-show-field-summary: False
|
||||
:members: coder_use_cache, data_folder, data_folder_debug, file_based_execution_timeout, select_method, max_loop, knowledge_base_path, new_knowledge_base_path
|
||||
:exclude-members: Config, fail_task_trial_limit, v1_query_former_trace_limit, v1_query_similar_success_limit, v2_query_component_limit, v2_query_error_limit, v2_query_former_trace_limit, v2_error_summary, v2_knowledge_sampler
|
||||
:no-index:
|
||||
|
||||
- **Qlib Configuration**
|
||||
- The `.yaml` files in both the `model_template` and `factor_template` directories contain some configurations for running the corresponding models or factors within the Qlib framework. Below is an overview of their contents and roles:
|
||||
- **General Settings**:
|
||||
- **provider_uri**: Specifies the local Qlib data path, set to `~/.qlib/qlib_data/cn_data`.
|
||||
- **market**: Configured to `csi300`, representing the CSI 300 index constituents.
|
||||
- **benchmark**: Set to `SH000300`, used for backtesting evaluation.
|
||||
|
||||
- **Data Handling**:
|
||||
- **start_time** and **end_time**: Define the full data range, from `2008-01-01` to `2022-08-01`.
|
||||
- **fit_start_time**: The start date for fitting the model, set to `2008-01-01`.
|
||||
- **fit_end_time**: The end date for fitting the model, set to `2014-12-31`.
|
||||
- **features and labels**: Generated via a nested data loader combining `Alpha158DL` (for engineered features such as `RESI5`, `WVMA5`, `RSQR5`, `KLEN`, etc.) and a `StaticDataLoader` that loads precomputed factor files (`combined_factors_df.parquet`).
|
||||
- **normalization**: The pipeline includes `RobustZScoreNorm` (with clipping) and `Fillna` for inference, and `DropnaLabel` with `CSZScoreNorm` for training.
|
||||
|
||||
- **Training Configuration**:
|
||||
- **Model**: Uses `GeneralPTNN`, a PyTorch-based neural network model.
|
||||
- **Dataset Splits**:
|
||||
- **train**: `2008-01-01` to `2014-12-31`
|
||||
- **valid**: `2015-01-01` to `2016-12-31`
|
||||
- **test**: `2017-01-01` to `2020-08-01`
|
||||
|
||||
- **Default Hyperparameters** (can be overridden by command-line arguments):
|
||||
- **n_epochs**: `100`
|
||||
- **lr**: `2e-4`
|
||||
- **early_stop**: `10`
|
||||
- **batch_size**: `256`
|
||||
- **weight_decay**: `0.0`
|
||||
- **metric**: `loss`
|
||||
- **loss**: `mse`
|
||||
- **n_jobs**: `20`
|
||||
- **GPU**: `0` (uses GPU 0 if available)
|
||||
|
||||
- **Backtesting and Evaluation**:
|
||||
- **strategy**: `TopkDropoutStrategy`, which selects the top 50 stocks and randomly drops 5 to introduce exploration.
|
||||
- **backtest period**: `2017-01-01` to `2020-08-01`
|
||||
- **initial capital**: `100,000,000`
|
||||
- **cost configuration**: Includes open/close costs, minimum transaction costs, and slippage control.
|
||||
|
||||
- **Recording and Analysis**:
|
||||
- **SignalRecord**: Logs predicted signals.
|
||||
- **SigAnaRecord**: Performs signal analysis without long-short separation.
|
||||
- **PortAnaRecord**: Conducts portfolio analysis using the configured strategy and backtest settings.
|
||||
@@ -38,6 +38,7 @@ Use Web App
|
||||
- Qlib Factor
|
||||
- Data Mining
|
||||
- Model from Paper
|
||||
- Kaggle
|
||||
|
||||
3. Click the `Config⚙️` button and input the log path (if you set the log_dir parameter, you can select a log_path in the dropdown list).
|
||||
|
||||
|
||||
+6
-2
@@ -61,6 +61,10 @@ explicit_package_bases = true
|
||||
warn_return_any = true
|
||||
warn_unused_ignores = true
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
ignore_missing_imports = true
|
||||
module = "llama"
|
||||
|
||||
[tool.pytest.ini_options]
|
||||
addopts = "-l -s --durations=0"
|
||||
log_cli = true
|
||||
@@ -77,10 +81,10 @@ src = ["rdagent"]
|
||||
[tool.ruff.lint]
|
||||
ignore = [
|
||||
# https://docs.astral.sh/ruff/rules/#pydocstyle-d
|
||||
"ANN101",
|
||||
"ANN401",
|
||||
"D",
|
||||
"ERA001",
|
||||
"EXE002",
|
||||
"FIX",
|
||||
"INP001",
|
||||
"PGH",
|
||||
@@ -88,7 +92,7 @@ ignore = [
|
||||
"S101",
|
||||
"S301",
|
||||
"T20",
|
||||
"TCH003",
|
||||
"TC003",
|
||||
"TD",
|
||||
]
|
||||
select = ["ALL"]
|
||||
|
||||
@@ -13,9 +13,10 @@ from rdagent.components.benchmark.eval_method import FactorImplementEval
|
||||
|
||||
|
||||
class BenchmarkAnalyzer:
|
||||
def __init__(self, settings):
|
||||
def __init__(self, settings, only_correct_format=False):
|
||||
self.settings = settings
|
||||
self.index_map = self.load_index_map()
|
||||
self.only_correct_format = only_correct_format
|
||||
|
||||
def load_index_map(self):
|
||||
index_map = {}
|
||||
@@ -82,12 +83,12 @@ class BenchmarkAnalyzer:
|
||||
for i in x:
|
||||
order_v.append(
|
||||
{
|
||||
"avg. Run successful rate": 0,
|
||||
"avg. Format successful rate": 1,
|
||||
"avg. Correlation (value only)": 2,
|
||||
"max. Correlation": 3,
|
||||
"max. accuracy": 4,
|
||||
"avg. accuracy": 5,
|
||||
"Avg Run SR": 0,
|
||||
"Avg Format SR": 1,
|
||||
"Avg Correlation": 2,
|
||||
"Max Correlation": 3,
|
||||
"Max Accuracy": 4,
|
||||
"Avg Accuracy": 5,
|
||||
}.get(i, i),
|
||||
)
|
||||
return order_v
|
||||
@@ -119,11 +120,13 @@ class BenchmarkAnalyzer:
|
||||
format_succ_rate_f = self.reformat_index(format_succ_rate)
|
||||
|
||||
corr = sum_df_clean["FactorCorrelationEvaluator"].fillna(0.0)
|
||||
corr = corr.unstack().T.mean(axis=0).to_frame("corr(only success)")
|
||||
corr_res = self.reformat_index(corr)
|
||||
corr_max = sum_df_clean["FactorCorrelationEvaluator"]
|
||||
if self.only_correct_format:
|
||||
corr = corr.loc[format_issue == 1.0]
|
||||
|
||||
corr_max = corr_max.unstack().T.max(axis=0).to_frame("corr(only success)")
|
||||
corr_res = corr.unstack().T.mean(axis=0).to_frame("corr(only success)")
|
||||
corr_res = self.reformat_index(corr_res)
|
||||
|
||||
corr_max = corr.unstack().T.max(axis=0).to_frame("corr(only success)")
|
||||
corr_max_res = self.reformat_index(corr_max)
|
||||
|
||||
value_max = sum_df_clean["FactorEqualValueRatioEvaluator"]
|
||||
@@ -140,19 +143,25 @@ class BenchmarkAnalyzer:
|
||||
|
||||
result_all = pd.concat(
|
||||
{
|
||||
"avg. Correlation (value only)": corr_res.iloc[:, 0],
|
||||
"avg. Format successful rate": format_succ_rate_f.iloc[:, 0],
|
||||
"avg. Run successful rate": succ_rate_f.iloc[:, 0],
|
||||
"max. Correlation": corr_max_res.iloc[:, 0],
|
||||
"max. accuracy": value_max_res.iloc[:, 0],
|
||||
"avg. accuracy": value_avg_res.iloc[:, 0],
|
||||
"Avg Correlation": corr_res.iloc[:, 0],
|
||||
"Avg Format SR": format_succ_rate_f.iloc[:, 0],
|
||||
"Avg Run SR": succ_rate_f.iloc[:, 0],
|
||||
"Max Correlation": corr_max_res.iloc[:, 0],
|
||||
"Max Accuracy": value_max_res.iloc[:, 0],
|
||||
"Avg Accuracy": value_avg_res.iloc[:, 0],
|
||||
},
|
||||
axis=1,
|
||||
)
|
||||
|
||||
df = result_all.sort_index(axis=1, key=self.result_all_key_order)
|
||||
df = result_all.sort_index(axis=1, key=self.result_all_key_order).sort_index(axis=0)
|
||||
print(df)
|
||||
|
||||
print()
|
||||
print(df.groupby("Category").mean())
|
||||
|
||||
print()
|
||||
print(df.mean())
|
||||
|
||||
# Calculate the mean of each column
|
||||
mean_values = df.fillna(0.0).mean()
|
||||
mean_df = pd.DataFrame(mean_values).T
|
||||
@@ -179,11 +188,16 @@ class Plotter:
|
||||
|
||||
@staticmethod
|
||||
def plot_data(data, file_name, title):
|
||||
plt.figure(figsize=(10, 6))
|
||||
sns.barplot(x="index", y="b", hue="a", data=data)
|
||||
plt.xlabel("Method")
|
||||
plt.figure(figsize=(10, 10))
|
||||
plt.ylabel("Value")
|
||||
plt.title(title)
|
||||
colors = ["#3274A1", "#E1812C", "#3A923A", "#C03D3E"]
|
||||
plt.bar(data["a"], data["b"], color=colors, capsize=5)
|
||||
for idx, row in data.iterrows():
|
||||
plt.text(idx, row["b"] + 0.01, f"{row['b']:.2f}", ha="center", va="bottom")
|
||||
plt.suptitle(title, y=0.98)
|
||||
plt.xticks(rotation=45)
|
||||
plt.ylim(0, 1)
|
||||
plt.tight_layout()
|
||||
plt.savefig(file_name)
|
||||
|
||||
|
||||
@@ -191,9 +205,10 @@ def main(
|
||||
path="git_ignore_folder/eval_results/res_promptV220240724-060037.pkl",
|
||||
round=1,
|
||||
title="Comparison of Different Methods",
|
||||
only_correct_format=False,
|
||||
):
|
||||
settings = BenchmarkSettings()
|
||||
benchmark = BenchmarkAnalyzer(settings)
|
||||
benchmark = BenchmarkAnalyzer(settings, only_correct_format=only_correct_format)
|
||||
results = {
|
||||
f"{round} round experiment": path,
|
||||
}
|
||||
@@ -201,7 +216,7 @@ def main(
|
||||
final_results_df = pd.DataFrame(final_results)
|
||||
|
||||
Plotter.change_fs(20)
|
||||
plot_data = final_results_df.drop(["max. accuracy", "avg. accuracy"], axis=0).T
|
||||
plot_data = final_results_df.drop(["Max Accuracy", "Avg Accuracy"], axis=0).T
|
||||
plot_data = plot_data.reset_index().melt("index", var_name="a", value_name="b")
|
||||
Plotter.plot_data(plot_data, "./comparison_plot.png", title)
|
||||
|
||||
|
||||
@@ -1,16 +1,9 @@
|
||||
import os
|
||||
import pickle
|
||||
import time
|
||||
from pathlib import Path
|
||||
from pprint import pprint
|
||||
|
||||
from rdagent.app.qlib_rd_loop.conf import FACTOR_PROP_SETTING
|
||||
from rdagent.components.benchmark.conf import BenchmarkSettings
|
||||
from rdagent.components.benchmark.eval_method import FactorImplementEval
|
||||
from rdagent.core.scenario import Scenario
|
||||
from rdagent.core.utils import import_class
|
||||
from rdagent.log import rdagent_logger as logger
|
||||
from rdagent.scenarios.qlib.experiment.factor_experiment import QlibFactorScenario
|
||||
from rdagent.scenarios.qlib.factor_experiment_loader.json_loader import (
|
||||
FactorTestCaseLoaderFromJsonFile,
|
||||
)
|
||||
@@ -25,7 +18,7 @@ if __name__ == "__main__":
|
||||
# 3.declare the method to be tested and pass the arguments.
|
||||
|
||||
scen: Scenario = import_class(FACTOR_PROP_SETTING.scen)()
|
||||
generate_method = import_class(bs.bench_method_cls)(scen=scen)
|
||||
generate_method = import_class(bs.bench_method_cls)(scen=scen, **bs.bench_method_extra_kwargs)
|
||||
# 4.declare the eval method and pass the arguments.
|
||||
eval_method = FactorImplementEval(
|
||||
method=generate_method,
|
||||
@@ -36,7 +29,7 @@ if __name__ == "__main__":
|
||||
)
|
||||
|
||||
# 5.run the eval
|
||||
res = eval_method.eval()
|
||||
res = eval_method.eval(eval_method.develop())
|
||||
|
||||
# 6.save the result
|
||||
logger.log_object(res)
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
from pathlib import Path
|
||||
|
||||
from rdagent.components.coder.model_coder.CoSTEER import ModelCoSTEER
|
||||
from rdagent.components.coder.model_coder import ModelCoSTEER
|
||||
from rdagent.components.loader.task_loader import ModelTaskLoaderJson, ModelWsLoader
|
||||
from rdagent.scenarios.qlib.experiment.model_experiment import (
|
||||
QlibModelExperiment,
|
||||
@@ -13,7 +13,7 @@ if __name__ == "__main__":
|
||||
from rdagent.components.coder.model_coder.benchmark.eval import ModelImpValEval
|
||||
from rdagent.components.coder.model_coder.one_shot import ModelCodeWriter
|
||||
|
||||
bench_folder = DIRNAME.parent.parent / "components" / "coder" / "model_coder" / "benchmark"
|
||||
bench_folder = DIRNAME.parent.parent.parent / "components" / "coder" / "model_coder" / "benchmark"
|
||||
mtl = ModelTaskLoaderJson(str(bench_folder / "model_dict.json"))
|
||||
|
||||
task_l = mtl.load()
|
||||
|
||||
+17
-6
@@ -1,10 +1,11 @@
|
||||
"""
|
||||
CLI entrance for all rdagent application.
|
||||
|
||||
This will
|
||||
This will
|
||||
- make rdagent a nice entry and
|
||||
- autoamtically load dotenv
|
||||
"""
|
||||
|
||||
from dotenv import load_dotenv
|
||||
|
||||
load_dotenv(".env")
|
||||
@@ -16,18 +17,19 @@ from importlib.resources import path as rpath
|
||||
|
||||
import fire
|
||||
|
||||
from rdagent.app.data_mining.model import main as med_model
|
||||
from rdagent.app.data_science.loop import main as data_science
|
||||
from rdagent.app.general_model.general_model import (
|
||||
extract_models_and_implement as general_model,
|
||||
)
|
||||
from rdagent.app.kaggle.loop import main as kaggle_main
|
||||
from rdagent.app.qlib_rd_loop.factor import main as fin_factor
|
||||
from rdagent.app.qlib_rd_loop.factor_from_report import main as fin_factor_report
|
||||
from rdagent.app.qlib_rd_loop.model import main as fin_model
|
||||
from rdagent.app.qlib_rd_loop.quant import main as fin_quant
|
||||
from rdagent.app.utils.health_check import health_check
|
||||
from rdagent.app.utils.info import collect_info
|
||||
|
||||
|
||||
def ui(port=80, log_dir="", debug=False):
|
||||
def ui(port=19899, log_dir="", debug=False):
|
||||
"""
|
||||
start web app to show the log traces.
|
||||
"""
|
||||
@@ -42,16 +44,25 @@ def ui(port=80, log_dir="", debug=False):
|
||||
subprocess.run(cmds)
|
||||
|
||||
|
||||
def server_ui(port=19899):
|
||||
"""
|
||||
start web app to show the log traces in real time
|
||||
"""
|
||||
subprocess.run(["python", "rdagent/log/server/app.py", f"--port={port}"])
|
||||
|
||||
|
||||
def app():
|
||||
fire.Fire(
|
||||
{
|
||||
"fin_factor": fin_factor,
|
||||
"fin_factor_report": fin_factor_report,
|
||||
"fin_model": fin_model,
|
||||
"med_model": med_model,
|
||||
"fin_quant": fin_quant,
|
||||
"general_model": general_model,
|
||||
"ui": ui,
|
||||
"health_check": health_check,
|
||||
"collect_info": collect_info,
|
||||
"kaggle": kaggle_main,
|
||||
"data_science": data_science,
|
||||
"server_ui": server_ui,
|
||||
}
|
||||
)
|
||||
|
||||
@@ -1,49 +0,0 @@
|
||||
from pathlib import Path
|
||||
|
||||
from pydantic_settings import BaseSettings
|
||||
|
||||
from rdagent.components.workflow.conf import BasePropSetting
|
||||
|
||||
|
||||
class MedBasePropSetting(BasePropSetting):
|
||||
class Config:
|
||||
env_prefix = "DM_"
|
||||
"""Use `DM_` as prefix for environment variables"""
|
||||
protected_namespaces = ()
|
||||
"""Add 'model_' to the protected namespaces"""
|
||||
|
||||
# 1) overriding the default
|
||||
scen: str = "rdagent.scenarios.data_mining.experiment.model_experiment.DMModelScenario"
|
||||
"""Scenario class for data mining model"""
|
||||
|
||||
hypothesis_gen: str = "rdagent.scenarios.data_mining.proposal.model_proposal.DMModelHypothesisGen"
|
||||
"""Hypothesis generation class"""
|
||||
|
||||
hypothesis2experiment: str = "rdagent.scenarios.data_mining.proposal.model_proposal.DMModelHypothesis2Experiment"
|
||||
"""Hypothesis to experiment class"""
|
||||
|
||||
coder: str = "rdagent.scenarios.data_mining.developer.model_coder.DMModelCoSTEER"
|
||||
"""Coder class"""
|
||||
|
||||
runner: str = "rdagent.scenarios.data_mining.developer.model_runner.DMModelRunner"
|
||||
"""Runner class"""
|
||||
|
||||
summarizer: str = "rdagent.scenarios.data_mining.developer.feedback.DMModelHypothesisExperiment2Feedback"
|
||||
"""Summarizer class"""
|
||||
|
||||
evolving_n: int = 10
|
||||
"""Number of evolutions"""
|
||||
|
||||
evolving_n: int = 10
|
||||
|
||||
# 2) Extra config for the scenario
|
||||
# physionet account
|
||||
# NOTE: You should apply the account in https://physionet.org/
|
||||
username: str = ""
|
||||
"""Physionet account username"""
|
||||
|
||||
password: str = ""
|
||||
"""Physionet account password"""
|
||||
|
||||
|
||||
MED_PROP_SETTING = MedBasePropSetting()
|
||||
@@ -1,31 +0,0 @@
|
||||
import fire
|
||||
|
||||
from rdagent.app.data_mining.conf import MED_PROP_SETTING
|
||||
from rdagent.components.workflow.rd_loop import RDLoop
|
||||
from rdagent.core.exception import ModelEmptyError
|
||||
|
||||
|
||||
class ModelRDLoop(RDLoop):
|
||||
skip_loop_error = (ModelEmptyError,)
|
||||
|
||||
|
||||
def main(path=None, step_n=None):
|
||||
"""
|
||||
Auto R&D Evolving loop for models in a medical scenario.
|
||||
|
||||
You can continue running session by
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
dotenv run -- python rdagent/app/data_mining/model.py $LOG_PATH/__session__/1/0_propose --step_n 1 # `step_n` is a optional paramter
|
||||
|
||||
"""
|
||||
if path is None:
|
||||
model_loop = ModelRDLoop(MED_PROP_SETTING)
|
||||
else:
|
||||
model_loop = ModelRDLoop.load(path)
|
||||
model_loop.run(step_n=step_n)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
fire.Fire(main)
|
||||
@@ -0,0 +1,115 @@
|
||||
from typing import Literal
|
||||
|
||||
from pydantic_settings import SettingsConfigDict
|
||||
|
||||
from rdagent.app.kaggle.conf import KaggleBasePropSetting
|
||||
|
||||
|
||||
class DataScienceBasePropSetting(KaggleBasePropSetting):
|
||||
# TODO: Kaggle Setting should be the subclass of DataScience
|
||||
model_config = SettingsConfigDict(env_prefix="DS_", protected_namespaces=())
|
||||
|
||||
# Main components
|
||||
## Scen
|
||||
scen: str = "rdagent.scenarios.data_science.scen.KaggleScen"
|
||||
"""
|
||||
Scenario class for data science tasks.
|
||||
- For Kaggle competitions, use: "rdagent.scenarios.data_science.scen.KaggleScen"
|
||||
- For custom data science scenarios, use: "rdagent.scenarios.data_science.scen.DataScienceScen"
|
||||
"""
|
||||
|
||||
hypothesis_gen: str = "rdagent.scenarios.data_science.proposal.exp_gen.proposal.DSProposalV2ExpGen"
|
||||
"""Hypothesis generation class"""
|
||||
|
||||
## Workflow Related
|
||||
consecutive_errors: int = 5
|
||||
|
||||
## Coding Related
|
||||
coding_fail_reanalyze_threshold: int = 3
|
||||
|
||||
debug_timeout: int = 600
|
||||
"""The timeout limit for running on debugging data"""
|
||||
full_timeout: int = 3600
|
||||
"""The timeout limit for running on full data"""
|
||||
|
||||
### specific feature
|
||||
|
||||
#### enable specification
|
||||
spec_enabled: bool = True
|
||||
|
||||
#### proposal related
|
||||
proposal_version: str = "v1"
|
||||
coder_on_whole_pipeline: bool = False
|
||||
max_trace_hist: int = 3
|
||||
|
||||
coder_max_loop: int = 10
|
||||
runner_max_loop: int = 1
|
||||
|
||||
rule_base_eval: bool = False
|
||||
sample_data: bool = True
|
||||
use_raw_description: bool = False
|
||||
show_nan_columns: bool = False
|
||||
|
||||
#### model dump
|
||||
enable_model_dump: bool = False
|
||||
enable_doc_dev: bool = False
|
||||
model_dump_check_level: Literal["medium", "high"] = "medium"
|
||||
|
||||
### knowledge base
|
||||
enable_knowledge_base: bool = False
|
||||
knowledge_base_version: str = "v1"
|
||||
knowledge_base_path: str | None = None
|
||||
idea_pool_json_path: str | None = None
|
||||
|
||||
### archive log folder after each loop
|
||||
enable_log_archive: bool = True
|
||||
log_archive_path: str | None = None
|
||||
log_archive_temp_path: str | None = (
|
||||
None # This is to store the mid tar file since writing the tar file is preferred in local storage then copy to target storage
|
||||
)
|
||||
|
||||
#### Evaluation on Test related
|
||||
eval_sub_dir: str = "eval" # TODO: fixme, this is not a good name
|
||||
"""We'll use f"{DS_RD_SETTING.local_data_path}/{DS_RD_SETTING.eval_sub_dir}/{competition}"
|
||||
to find the scriipt to evaluate the submission on test"""
|
||||
|
||||
"""---below are the settings for multi-trace---"""
|
||||
|
||||
### multi-trace related
|
||||
max_trace_num: int = 3
|
||||
"""The maximum number of traces to grow before merging"""
|
||||
|
||||
#### multi-trace:checkpoint selector
|
||||
selector_name: str = "rdagent.scenarios.data_science.proposal.exp_gen.ckp_select.LatestCKPSelector"
|
||||
"""The name of the selector to use"""
|
||||
sota_count_window: int = 5
|
||||
"""The number of trials to consider for SOTA count"""
|
||||
sota_count_threshold: int = 1
|
||||
"""The threshold for SOTA count"""
|
||||
|
||||
#### multi-trace: SOTA experiment selector
|
||||
sota_exp_selector_name: str = "rdagent.scenarios.data_science.proposal.exp_gen.sota_exp_select.GlobalSOTASelector"
|
||||
"""The name of the SOTA experiment selector to use"""
|
||||
|
||||
### multi-trace:inject optimals for multi-trace
|
||||
# inject diverse when start a new sub-trace
|
||||
enable_inject_diverse: bool = False
|
||||
|
||||
# inject knowledge at the root of the trace
|
||||
enable_inject_knowledge_at_root: bool = False
|
||||
|
||||
# enable different version of DSExpGen for multi-trace
|
||||
enable_multi_version_exp_gen: bool = False
|
||||
exp_gen_version_list: str = "v3,v2"
|
||||
|
||||
#### multi-trace: time for final multi-trace merge
|
||||
merge_hours: int = 2
|
||||
"""The time for merge"""
|
||||
|
||||
#### multi-trace: max SOTA-retrieved number, used in AutoSOTAexpSelector
|
||||
# constrains the number of SOTA experiments to retrieve, otherwise too many SOTA experiments to retrieve will cause the exceed of the context window of LLM
|
||||
max_sota_retrieved_num: int = 10
|
||||
"""The maximum number of SOTA experiments to retrieve in a LLM call"""
|
||||
|
||||
|
||||
DS_RD_SETTING = DataScienceBasePropSetting()
|
||||
@@ -0,0 +1,6 @@
|
||||
import fire
|
||||
|
||||
from rdagent.scenarios.data_science.debug.data import create_debug_data
|
||||
|
||||
if __name__ == "__main__":
|
||||
fire.Fire(create_debug_data)
|
||||
@@ -0,0 +1,74 @@
|
||||
import asyncio
|
||||
from pathlib import Path
|
||||
|
||||
import fire
|
||||
|
||||
from rdagent.app.data_science.conf import DS_RD_SETTING
|
||||
from rdagent.core.utils import import_class
|
||||
from rdagent.log import rdagent_logger as logger
|
||||
from rdagent.scenarios.data_science.loop import DataScienceRDLoop
|
||||
|
||||
|
||||
def main(
|
||||
path: str | None = None,
|
||||
checkout: bool | str | Path = True,
|
||||
step_n: int | None = None,
|
||||
loop_n: int | None = None,
|
||||
competition="bms-molecular-translation",
|
||||
timeout=None,
|
||||
replace_timer=True,
|
||||
exp_gen_cls: str | None = None,
|
||||
):
|
||||
"""
|
||||
|
||||
Parameters
|
||||
----------
|
||||
path :
|
||||
A path like `$LOG_PATH/__session__/1/0_propose`. This indicates that we restore the state after finishing step 0 in loop 1.
|
||||
checkout :
|
||||
Used only when a path is provided.
|
||||
Can be True, False, or a path.
|
||||
Default is True.
|
||||
- If True, the new loop will use the existing folder and clear logs for sessions after the one corresponding to the given path.
|
||||
- If False, the new loop will use the existing folder but keep the logs for sessions after the one corresponding to the given path.
|
||||
- If a path (or a str like Path) is provided, the new loop will be saved to that path, leaving the original path unchanged.
|
||||
step_n :
|
||||
Number of steps to run; if None, the process will run indefinitely until an error or KeyboardInterrupt occurs.
|
||||
loop_n :
|
||||
Number of loops to run; if None, the process will run indefinitely until an error or KeyboardInterrupt occurs.
|
||||
- If the current loop is incomplete, it will be counted as the first loop for completion.
|
||||
- If both step_n and loop_n are provided, the process will stop as soon as either condition is met.
|
||||
competition :
|
||||
Competition name.
|
||||
replace_timer :
|
||||
If a session is loaded, determines whether to replace the timer with session.timer.
|
||||
exp_gen_cls :
|
||||
When there are different stages, the exp_gen can be replaced with the new proposal.
|
||||
|
||||
|
||||
Auto R&D Evolving loop for models in a Kaggle scenario.
|
||||
You can continue running a session by using the command:
|
||||
.. code-block:: bash
|
||||
dotenv run -- python rdagent/app/data_science/loop.py [--competition titanic] $LOG_PATH/__session__/1/0_propose --step_n 1 # `step_n` is an optional parameter
|
||||
rdagent kaggle --competition playground-series-s4e8 # This command is recommended.
|
||||
"""
|
||||
if competition is not None:
|
||||
DS_RD_SETTING.competition = competition
|
||||
|
||||
if not DS_RD_SETTING.competition:
|
||||
logger.error("Please specify competition name.")
|
||||
|
||||
if path is None:
|
||||
kaggle_loop = DataScienceRDLoop(DS_RD_SETTING)
|
||||
else:
|
||||
kaggle_loop: DataScienceRDLoop = DataScienceRDLoop.load(path, checkout=checkout, replace_timer=replace_timer)
|
||||
|
||||
# replace exp_gen if we have new class
|
||||
if exp_gen_cls is not None:
|
||||
kaggle_loop.exp_gen = import_class(exp_gen_cls)(kaggle_loop.exp_gen.scen)
|
||||
|
||||
asyncio.run(kaggle_loop.run(step_n=step_n, loop_n=loop_n, all_duration=timeout))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
fire.Fire(main)
|
||||
@@ -30,19 +30,15 @@ def extract_models_and_implement(report_file_path: str) -> None:
|
||||
Returns:
|
||||
None
|
||||
"""
|
||||
with logger.tag("init"):
|
||||
scenario = GeneralModelScenario()
|
||||
logger.log_object(scenario, tag="scenario")
|
||||
with logger.tag("r"):
|
||||
# Save Relevant Images
|
||||
img = extract_first_page_screenshot_from_pdf(report_file_path)
|
||||
logger.log_object(img, tag="pdf_image")
|
||||
exp = ModelExperimentLoaderFromPDFfiles().load(report_file_path)
|
||||
logger.log_object(exp, tag="load_experiment")
|
||||
with logger.tag("d"):
|
||||
exp = QlibModelCoSTEER(scenario).develop(exp)
|
||||
logger.log_object(exp, tag="developed_experiment")
|
||||
return exp
|
||||
scenario = GeneralModelScenario()
|
||||
logger.log_object(scenario, tag="scenario")
|
||||
# Save Relevant Images
|
||||
img = extract_first_page_screenshot_from_pdf(report_file_path)
|
||||
logger.log_object(img, tag="pdf_image")
|
||||
exp = ModelExperimentLoaderFromPDFfiles().load(report_file_path)
|
||||
logger.log_object(exp, tag="load_experiment")
|
||||
exp = QlibModelCoSTEER(scenario).develop(exp)
|
||||
logger.log_object(exp, tag="developed_experiment")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
+31
-24
@@ -1,28 +1,15 @@
|
||||
from pathlib import Path
|
||||
from pydantic_settings import SettingsConfigDict
|
||||
|
||||
from pydantic_settings import BaseSettings
|
||||
|
||||
from rdagent.components.workflow.conf import BasePropSetting
|
||||
from rdagent.core.conf import ExtendedBaseSettings
|
||||
|
||||
|
||||
class KaggleBasePropSetting(BasePropSetting):
|
||||
class Config:
|
||||
env_prefix = "KG_"
|
||||
"""Use `KG_` as prefix for environment variables"""
|
||||
protected_namespaces = ()
|
||||
"""Do not allow overriding of these namespaces"""
|
||||
class KaggleBasePropSetting(ExtendedBaseSettings):
|
||||
model_config = SettingsConfigDict(env_prefix="KG_", protected_namespaces=())
|
||||
|
||||
# 1) overriding the default
|
||||
scen: str = "rdagent.scenarios.kaggle.experiment.scenario.KGScenario"
|
||||
"""Scenario class for data mining model"""
|
||||
|
||||
knowledge_base: str = "" # TODO enable this line to use the knowledge base
|
||||
# knowledge_base: str = "rdagent.scenarios.kaggle.knowledge_management.graph.KGKnowledgeGraph"
|
||||
"""Knowledge base class"""
|
||||
|
||||
knowledge_base_path: str = "kg_graph.pkl"
|
||||
"""Knowledge base path"""
|
||||
|
||||
hypothesis_gen: str = "rdagent.scenarios.kaggle.proposal.proposal.KGHypothesisGen"
|
||||
"""Hypothesis generation class"""
|
||||
|
||||
@@ -44,29 +31,49 @@ class KaggleBasePropSetting(BasePropSetting):
|
||||
model_runner: str = "rdagent.scenarios.kaggle.developer.runner.KGModelRunner"
|
||||
"""Model Runner class"""
|
||||
|
||||
summarizer: str = "rdagent.scenarios.kaggle.developer.feedback.KGHypothesisExperiment2Feedback"
|
||||
summarizer: str = "rdagent.scenarios.kaggle.developer.feedback.KGExperiment2Feedback"
|
||||
"""Summarizer class"""
|
||||
|
||||
evolving_n: int = 10
|
||||
"""Number of evolutions"""
|
||||
|
||||
competition: str = ""
|
||||
"""Kaggle competition name, e.g., 'sf-crime'"""
|
||||
|
||||
local_data_path: str = "/data/userdata/share/kaggle"
|
||||
template_path: str = "rdagent/scenarios/kaggle/experiment/templates"
|
||||
"""Kaggle competition base templates path"""
|
||||
|
||||
local_data_path: str = ""
|
||||
"""Folder storing Kaggle competition data"""
|
||||
|
||||
# Evaluation on Test related
|
||||
if_using_mle_data: bool = False
|
||||
auto_submit: bool = False
|
||||
"""Automatically upload and submit each experiment result to Kaggle platform"""
|
||||
|
||||
# Conditionally set the knowledge_base based on the use of graph RAG
|
||||
knowledge_base: str = ""
|
||||
"""Knowledge base class, uses 'KGKnowledgeGraph' when advanced graph-based RAG is enabled, otherwise empty."""
|
||||
if_action_choosing_based_on_UCB: bool = False
|
||||
"""Enable decision mechanism based on UCB algorithm"""
|
||||
|
||||
domain_knowledge_path: str = "/data/userdata/share/kaggle/domain_knowledge"
|
||||
"""Folder storing domain knowledge files in .case format"""
|
||||
|
||||
rag_path: str = "git_ignore_folder/rag"
|
||||
knowledge_base_path: str = "kg_graph.pkl"
|
||||
"""Advanced version of graph-based RAG"""
|
||||
|
||||
if_action_choosing_based_on_UCB: bool = False
|
||||
|
||||
if_using_graph_rag: bool = False
|
||||
rag_path: str = "git_ignore_folder/kaggle_vector_base.pkl"
|
||||
"""Base version of vector-based RAG"""
|
||||
|
||||
if_using_vector_rag: bool = False
|
||||
"""Enable basic vector-based RAG"""
|
||||
|
||||
auto_submit: bool = True
|
||||
if_using_graph_rag: bool = False
|
||||
"""Enable advanced graph-based RAG"""
|
||||
|
||||
mini_case: bool = False
|
||||
"""Enable mini-case study for experiments"""
|
||||
|
||||
|
||||
KAGGLE_IMPLEMENT_SETTING = KaggleBasePropSetting()
|
||||
|
||||
+80
-83
@@ -1,5 +1,4 @@
|
||||
import subprocess
|
||||
from collections import defaultdict
|
||||
from typing import Any
|
||||
|
||||
import fire
|
||||
@@ -8,17 +7,15 @@ from rdagent.app.kaggle.conf import KAGGLE_IMPLEMENT_SETTING
|
||||
from rdagent.components.workflow.conf import BasePropSetting
|
||||
from rdagent.components.workflow.rd_loop import RDLoop
|
||||
from rdagent.core.developer import Developer
|
||||
from rdagent.core.exception import FactorEmptyError, ModelEmptyError
|
||||
from rdagent.core.exception import CoderError, FactorEmptyError, ModelEmptyError
|
||||
from rdagent.core.proposal import (
|
||||
Experiment2Feedback,
|
||||
Hypothesis2Experiment,
|
||||
HypothesisExperiment2Feedback,
|
||||
HypothesisGen,
|
||||
Trace,
|
||||
)
|
||||
from rdagent.core.scenario import Scenario
|
||||
from rdagent.core.utils import import_class
|
||||
from rdagent.log import rdagent_logger as logger
|
||||
from rdagent.log.time import measure_time
|
||||
from rdagent.scenarios.kaggle.experiment.scenario import (
|
||||
KG_ACTION_FEATURE_ENGINEERING,
|
||||
KG_ACTION_FEATURE_PROCESSING,
|
||||
@@ -30,92 +27,88 @@ from rdagent.scenarios.kaggle.proposal.proposal import KGTrace
|
||||
|
||||
|
||||
class KaggleRDLoop(RDLoop):
|
||||
@measure_time
|
||||
def __init__(self, PROP_SETTING: BasePropSetting):
|
||||
with logger.tag("init"):
|
||||
scen: Scenario = import_class(PROP_SETTING.scen)(PROP_SETTING.competition)
|
||||
logger.log_object(scen, tag="scenario")
|
||||
knowledge_base = (
|
||||
import_class(PROP_SETTING.knowledge_base)(PROP_SETTING.knowledge_base_path, scen)
|
||||
if PROP_SETTING.knowledge_base != ""
|
||||
else None
|
||||
)
|
||||
logger.log_object(knowledge_base, tag="knowledge_base")
|
||||
self.hypothesis_gen: HypothesisGen = import_class(PROP_SETTING.hypothesis_gen)(scen)
|
||||
logger.log_object(self.hypothesis_gen, tag="hypothesis generator")
|
||||
self.hypothesis2experiment: Hypothesis2Experiment = import_class(PROP_SETTING.hypothesis2experiment)()
|
||||
logger.log_object(self.hypothesis2experiment, tag="hypothesis2experiment")
|
||||
self.feature_coder: Developer = import_class(PROP_SETTING.feature_coder)(scen)
|
||||
logger.log_object(self.feature_coder, tag="feature coder")
|
||||
self.model_feature_selection_coder: Developer = import_class(PROP_SETTING.model_feature_selection_coder)(
|
||||
scen
|
||||
)
|
||||
logger.log_object(self.model_feature_selection_coder, tag="model feature selection coder")
|
||||
self.model_coder: Developer = import_class(PROP_SETTING.model_coder)(scen)
|
||||
logger.log_object(self.model_coder, tag="model coder")
|
||||
self.feature_runner: Developer = import_class(PROP_SETTING.feature_runner)(scen)
|
||||
logger.log_object(self.feature_runner, tag="feature runner")
|
||||
self.model_runner: Developer = import_class(PROP_SETTING.model_runner)(scen)
|
||||
logger.log_object(self.model_runner, tag="model runner")
|
||||
self.summarizer: HypothesisExperiment2Feedback = import_class(PROP_SETTING.summarizer)(scen)
|
||||
logger.log_object(self.summarizer, tag="summarizer")
|
||||
self.trace = KGTrace(scen=scen, knowledge_base=knowledge_base)
|
||||
super(RDLoop, self).__init__()
|
||||
scen: Scenario = import_class(PROP_SETTING.scen)(PROP_SETTING.competition)
|
||||
logger.log_object(scen, tag="scenario")
|
||||
knowledge_base = (
|
||||
import_class(PROP_SETTING.knowledge_base)(PROP_SETTING.knowledge_base_path, scen)
|
||||
if PROP_SETTING.knowledge_base != ""
|
||||
else None
|
||||
)
|
||||
logger.log_object(knowledge_base, tag="knowledge_base")
|
||||
self.hypothesis_gen: HypothesisGen = import_class(PROP_SETTING.hypothesis_gen)(scen)
|
||||
logger.log_object(self.hypothesis_gen, tag="hypothesis generator")
|
||||
self.hypothesis2experiment: Hypothesis2Experiment = import_class(PROP_SETTING.hypothesis2experiment)()
|
||||
logger.log_object(self.hypothesis2experiment, tag="hypothesis2experiment")
|
||||
self.feature_coder: Developer = import_class(PROP_SETTING.feature_coder)(scen)
|
||||
logger.log_object(self.feature_coder, tag="feature coder")
|
||||
self.model_feature_selection_coder: Developer = import_class(PROP_SETTING.model_feature_selection_coder)(scen)
|
||||
logger.log_object(self.model_feature_selection_coder, tag="model feature selection coder")
|
||||
self.model_coder: Developer = import_class(PROP_SETTING.model_coder)(scen)
|
||||
logger.log_object(self.model_coder, tag="model coder")
|
||||
self.feature_runner: Developer = import_class(PROP_SETTING.feature_runner)(scen)
|
||||
logger.log_object(self.feature_runner, tag="feature runner")
|
||||
self.model_runner: Developer = import_class(PROP_SETTING.model_runner)(scen)
|
||||
logger.log_object(self.model_runner, tag="model runner")
|
||||
self.summarizer: Experiment2Feedback = import_class(PROP_SETTING.summarizer)(scen)
|
||||
logger.log_object(self.summarizer, tag="summarizer")
|
||||
self.trace = KGTrace(scen=scen, knowledge_base=knowledge_base)
|
||||
super(RDLoop, self).__init__()
|
||||
|
||||
@measure_time
|
||||
def coding(self, prev_out: dict[str, Any]):
|
||||
with logger.tag("d"): # develop
|
||||
if prev_out["propose"].action in [KG_ACTION_FEATURE_ENGINEERING, KG_ACTION_FEATURE_PROCESSING]:
|
||||
exp = self.feature_coder.develop(prev_out["exp_gen"])
|
||||
elif prev_out["propose"].action == KG_ACTION_MODEL_FEATURE_SELECTION:
|
||||
exp = self.model_feature_selection_coder.develop(prev_out["exp_gen"])
|
||||
else:
|
||||
exp = self.model_coder.develop(prev_out["exp_gen"])
|
||||
logger.log_object(exp.sub_workspace_list, tag="coder result")
|
||||
if prev_out["direct_exp_gen"]["propose"].action in [
|
||||
KG_ACTION_FEATURE_ENGINEERING,
|
||||
KG_ACTION_FEATURE_PROCESSING,
|
||||
]:
|
||||
exp = self.feature_coder.develop(prev_out["direct_exp_gen"]["exp_gen"])
|
||||
elif prev_out["direct_exp_gen"]["propose"].action == KG_ACTION_MODEL_FEATURE_SELECTION:
|
||||
exp = self.model_feature_selection_coder.develop(prev_out["direct_exp_gen"]["exp_gen"])
|
||||
else:
|
||||
exp = self.model_coder.develop(prev_out["direct_exp_gen"]["exp_gen"])
|
||||
logger.log_object(exp.sub_workspace_list, tag="coder result")
|
||||
return exp
|
||||
|
||||
@measure_time
|
||||
def running(self, prev_out: dict[str, Any]):
|
||||
with logger.tag("ef"): # evaluate and feedback
|
||||
if prev_out["propose"].action in [KG_ACTION_FEATURE_ENGINEERING, KG_ACTION_FEATURE_PROCESSING]:
|
||||
exp = self.feature_runner.develop(prev_out["coding"])
|
||||
else:
|
||||
exp = self.model_runner.develop(prev_out["coding"])
|
||||
logger.log_object(exp, tag="runner result")
|
||||
if KAGGLE_IMPLEMENT_SETTING.competition in [
|
||||
"optiver-realized-volatility-prediction",
|
||||
"covid19-global-forecasting-week-1",
|
||||
]:
|
||||
try:
|
||||
python_files_to_notebook(
|
||||
KAGGLE_IMPLEMENT_SETTING.competition, exp.experiment_workspace.workspace_path
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error(f"Merge python files to one file failed: {e}")
|
||||
if KAGGLE_IMPLEMENT_SETTING.auto_submit:
|
||||
csv_path = exp.experiment_workspace.workspace_path / "submission.csv"
|
||||
try:
|
||||
subprocess.run(
|
||||
[
|
||||
"kaggle",
|
||||
"competitions",
|
||||
"submit",
|
||||
"-f",
|
||||
str(csv_path.absolute()),
|
||||
"-m",
|
||||
str(csv_path.parent.absolute()),
|
||||
KAGGLE_IMPLEMENT_SETTING.competition,
|
||||
],
|
||||
check=True,
|
||||
)
|
||||
except subprocess.CalledProcessError as e:
|
||||
logger.error(f"Auto submission failed: \n{e}")
|
||||
except Exception as e:
|
||||
logger.error(f"Other exception when use kaggle api:\n{e}")
|
||||
if prev_out["direct_exp_gen"]["propose"].action in [
|
||||
KG_ACTION_FEATURE_ENGINEERING,
|
||||
KG_ACTION_FEATURE_PROCESSING,
|
||||
]:
|
||||
exp = self.feature_runner.develop(prev_out["coding"])
|
||||
else:
|
||||
exp = self.model_runner.develop(prev_out["coding"])
|
||||
logger.log_object(exp, tag="runner result")
|
||||
if KAGGLE_IMPLEMENT_SETTING.competition in [
|
||||
"optiver-realized-volatility-prediction",
|
||||
"covid19-global-forecasting-week-1",
|
||||
]:
|
||||
try:
|
||||
python_files_to_notebook(KAGGLE_IMPLEMENT_SETTING.competition, exp.experiment_workspace.workspace_path)
|
||||
except Exception as e:
|
||||
logger.error(f"Merge python files to one file failed: {e}")
|
||||
if KAGGLE_IMPLEMENT_SETTING.auto_submit:
|
||||
csv_path = exp.experiment_workspace.workspace_path / "submission.csv"
|
||||
try:
|
||||
subprocess.run(
|
||||
[
|
||||
"kaggle",
|
||||
"competitions",
|
||||
"submit",
|
||||
"-f",
|
||||
str(csv_path.absolute()),
|
||||
"-m",
|
||||
str(csv_path.parent.absolute()),
|
||||
KAGGLE_IMPLEMENT_SETTING.competition,
|
||||
],
|
||||
check=True,
|
||||
)
|
||||
except subprocess.CalledProcessError as e:
|
||||
logger.error(f"Auto submission failed: \n{e}")
|
||||
except Exception as e:
|
||||
logger.error(f"Other exception when use kaggle api:\n{e}")
|
||||
|
||||
return exp
|
||||
|
||||
skip_loop_error = (ModelEmptyError, FactorEmptyError)
|
||||
skip_loop_error = (ModelEmptyError, FactorEmptyError, CoderError)
|
||||
|
||||
|
||||
def main(path=None, step_n=None, competition=None):
|
||||
@@ -128,7 +121,11 @@ def main(path=None, step_n=None, competition=None):
|
||||
"""
|
||||
if competition:
|
||||
KAGGLE_IMPLEMENT_SETTING.competition = competition
|
||||
download_data(competition=competition, local_path=KAGGLE_IMPLEMENT_SETTING.local_data_path)
|
||||
download_data(competition=competition, settings=KAGGLE_IMPLEMENT_SETTING)
|
||||
if KAGGLE_IMPLEMENT_SETTING.if_using_graph_rag:
|
||||
KAGGLE_IMPLEMENT_SETTING.knowledge_base = (
|
||||
"rdagent.scenarios.kaggle.knowledge_management.graph.KGKnowledgeGraph"
|
||||
)
|
||||
else:
|
||||
logger.error("Please specify competition name.")
|
||||
if path is None:
|
||||
|
||||
@@ -1,14 +1,10 @@
|
||||
from pydantic_settings import BaseSettings
|
||||
from pydantic_settings import SettingsConfigDict
|
||||
|
||||
from rdagent.components.workflow.conf import BasePropSetting
|
||||
|
||||
|
||||
class ModelBasePropSetting(BasePropSetting):
|
||||
class Config:
|
||||
env_prefix = "QLIB_MODEL_"
|
||||
"""Use `QLIB_MODEL_` as prefix for environment variables"""
|
||||
protected_namespaces = ()
|
||||
"""Add 'model_' to the protected namespaces"""
|
||||
model_config = SettingsConfigDict(env_prefix="QLIB_MODEL_", protected_namespaces=())
|
||||
|
||||
# 1) override base settings
|
||||
scen: str = "rdagent.scenarios.qlib.experiment.model_experiment.QlibModelScenario"
|
||||
@@ -26,7 +22,7 @@ class ModelBasePropSetting(BasePropSetting):
|
||||
runner: str = "rdagent.scenarios.qlib.developer.model_runner.QlibModelRunner"
|
||||
"""Runner class"""
|
||||
|
||||
summarizer: str = "rdagent.scenarios.qlib.developer.feedback.QlibModelHypothesisExperiment2Feedback"
|
||||
summarizer: str = "rdagent.scenarios.qlib.developer.feedback.QlibModelExperiment2Feedback"
|
||||
"""Summarizer class"""
|
||||
|
||||
evolving_n: int = 10
|
||||
@@ -34,11 +30,7 @@ class ModelBasePropSetting(BasePropSetting):
|
||||
|
||||
|
||||
class FactorBasePropSetting(BasePropSetting):
|
||||
class Config:
|
||||
env_prefix = "QLIB_FACTOR_"
|
||||
"""Use `QLIB_FACTOR_` as prefix for environment variables"""
|
||||
protected_namespaces = ()
|
||||
"""Add 'factor_' to the protected namespaces"""
|
||||
model_config = SettingsConfigDict(env_prefix="QLIB_FACTOR_", protected_namespaces=())
|
||||
|
||||
# 1) override base settings
|
||||
scen: str = "rdagent.scenarios.qlib.experiment.factor_experiment.QlibFactorScenario"
|
||||
@@ -56,7 +48,7 @@ class FactorBasePropSetting(BasePropSetting):
|
||||
runner: str = "rdagent.scenarios.qlib.developer.factor_runner.QlibFactorRunner"
|
||||
"""Runner class"""
|
||||
|
||||
summarizer: str = "rdagent.scenarios.qlib.developer.feedback.QlibFactorHypothesisExperiment2Feedback"
|
||||
summarizer: str = "rdagent.scenarios.qlib.developer.feedback.QlibFactorExperiment2Feedback"
|
||||
"""Summarizer class"""
|
||||
|
||||
evolving_n: int = 10
|
||||
@@ -75,7 +67,54 @@ class FactorFromReportPropSetting(FactorBasePropSetting):
|
||||
max_factors_per_exp: int = 10000
|
||||
"""Maximum number of factors implemented per experiment"""
|
||||
|
||||
report_limit: int = 10000
|
||||
"""Maximum number of reports to process"""
|
||||
|
||||
|
||||
class QuantBasePropSetting(BasePropSetting):
|
||||
model_config = SettingsConfigDict(env_prefix="QLIB_QUANT_", protected_namespaces=())
|
||||
|
||||
# 1) override base settings
|
||||
scen: str = "rdagent.scenarios.qlib.experiment.quant_experiment.QlibQuantScenario"
|
||||
"""Scenario class for Qlib Model"""
|
||||
|
||||
quant_hypothesis_gen: str = "rdagent.scenarios.qlib.proposal.quant_proposal.QlibQuantHypothesisGen"
|
||||
"""Hypothesis generation class"""
|
||||
|
||||
model_hypothesis2experiment: str = "rdagent.scenarios.qlib.proposal.model_proposal.QlibModelHypothesis2Experiment"
|
||||
"""Hypothesis to experiment class"""
|
||||
|
||||
model_coder: str = "rdagent.scenarios.qlib.developer.model_coder.QlibModelCoSTEER"
|
||||
"""Coder class"""
|
||||
|
||||
model_runner: str = "rdagent.scenarios.qlib.developer.model_runner.QlibModelRunner"
|
||||
"""Runner class"""
|
||||
|
||||
model_summarizer: str = "rdagent.scenarios.qlib.developer.feedback.QlibModelExperiment2Feedback"
|
||||
"""Summarizer class"""
|
||||
|
||||
factor_hypothesis2experiment: str = (
|
||||
"rdagent.scenarios.qlib.proposal.factor_proposal.QlibFactorHypothesis2Experiment"
|
||||
)
|
||||
"""Hypothesis to experiment class"""
|
||||
|
||||
factor_coder: str = "rdagent.scenarios.qlib.developer.factor_coder.QlibFactorCoSTEER"
|
||||
"""Coder class"""
|
||||
|
||||
factor_runner: str = "rdagent.scenarios.qlib.developer.factor_runner.QlibFactorRunner"
|
||||
"""Runner class"""
|
||||
|
||||
factor_summarizer: str = "rdagent.scenarios.qlib.developer.feedback.QlibFactorExperiment2Feedback"
|
||||
"""Summarizer class"""
|
||||
|
||||
evolving_n: int = 10
|
||||
"""Number of evolutions"""
|
||||
|
||||
action_selection: str = "bandit"
|
||||
"""Action selection strategy: 'bandit' for bandit-based selection, 'llm' for LLM-based selection, 'random' for random selection"""
|
||||
|
||||
|
||||
FACTOR_PROP_SETTING = FactorBasePropSetting()
|
||||
FACTOR_FROM_REPORT_PROP_SETTING = FactorFromReportPropSetting()
|
||||
MODEL_PROP_SETTING = ModelBasePropSetting()
|
||||
QUANT_PROP_SETTING = QuantBasePropSetting()
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
Factor workflow with session control
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
from typing import Any
|
||||
|
||||
import fire
|
||||
@@ -10,24 +11,21 @@ from rdagent.app.qlib_rd_loop.conf import FACTOR_PROP_SETTING
|
||||
from rdagent.components.workflow.rd_loop import RDLoop
|
||||
from rdagent.core.exception import FactorEmptyError
|
||||
from rdagent.log import rdagent_logger as logger
|
||||
from rdagent.log.time import measure_time
|
||||
|
||||
|
||||
class FactorRDLoop(RDLoop):
|
||||
skip_loop_error = (FactorEmptyError,)
|
||||
|
||||
@measure_time
|
||||
def running(self, prev_out: dict[str, Any]):
|
||||
with logger.tag("ef"): # evaluate and feedback
|
||||
exp = self.runner.develop(prev_out["coding"])
|
||||
if exp is None:
|
||||
logger.error(f"Factor extraction failed.")
|
||||
raise FactorEmptyError("Factor extraction failed.")
|
||||
logger.log_object(exp, tag="runner result")
|
||||
exp = self.runner.develop(prev_out["coding"])
|
||||
if exp is None:
|
||||
logger.error(f"Factor extraction failed.")
|
||||
raise FactorEmptyError("Factor extraction failed.")
|
||||
logger.log_object(exp, tag="runner result")
|
||||
return exp
|
||||
|
||||
|
||||
def main(path=None, step_n=None):
|
||||
def main(path=None, step_n=None, loop_n=None, all_duration=None, checkout=True):
|
||||
"""
|
||||
Auto R&D Evolving loop for fintech factors.
|
||||
|
||||
@@ -41,8 +39,8 @@ def main(path=None, step_n=None):
|
||||
if path is None:
|
||||
model_loop = FactorRDLoop(FACTOR_PROP_SETTING)
|
||||
else:
|
||||
model_loop = FactorRDLoop.load(path)
|
||||
model_loop.run(step_n=step_n)
|
||||
model_loop = FactorRDLoop.load(path, checkout=checkout)
|
||||
asyncio.run(model_loop.run(step_n=step_n, loop_n=loop_n, all_duration=all_duration))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
import asyncio
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any, Tuple
|
||||
from typing import Any, Dict, Tuple
|
||||
|
||||
import fire
|
||||
from jinja2 import Environment, StrictUndefined
|
||||
|
||||
from rdagent.app.qlib_rd_loop.conf import FACTOR_FROM_REPORT_PROP_SETTING
|
||||
from rdagent.app.qlib_rd_loop.factor import FactorRDLoop
|
||||
@@ -11,20 +11,16 @@ from rdagent.components.document_reader.document_reader import (
|
||||
extract_first_page_screenshot_from_pdf,
|
||||
load_and_process_pdfs_by_langchain,
|
||||
)
|
||||
from rdagent.core.prompts import Prompts
|
||||
from rdagent.core.proposal import Hypothesis
|
||||
from rdagent.log import rdagent_logger as logger
|
||||
from rdagent.log.time import measure_time
|
||||
from rdagent.oai.llm_utils import APIBackend
|
||||
from rdagent.scenarios.qlib.experiment.factor_experiment import QlibFactorExperiment
|
||||
from rdagent.scenarios.qlib.factor_experiment_loader.pdf_loader import (
|
||||
FactorExperimentLoaderFromPDFfiles,
|
||||
)
|
||||
from rdagent.utils.agent.tpl import T
|
||||
from rdagent.utils.workflow import LoopMeta
|
||||
|
||||
prompts_path = Path(__file__).parent / "prompts.yaml"
|
||||
prompts = Prompts(file_path=prompts_path)
|
||||
|
||||
|
||||
def generate_hypothesis(factor_result: dict, report_content: str) -> str:
|
||||
"""
|
||||
@@ -37,19 +33,16 @@ def generate_hypothesis(factor_result: dict, report_content: str) -> str:
|
||||
Returns:
|
||||
str: The generated hypothesis.
|
||||
"""
|
||||
system_prompt = (
|
||||
Environment(undefined=StrictUndefined).from_string(prompts["hypothesis_generation"]["system"]).render()
|
||||
)
|
||||
user_prompt = (
|
||||
Environment(undefined=StrictUndefined)
|
||||
.from_string(prompts["hypothesis_generation"]["user"])
|
||||
.render(factor_descriptions=json.dumps(factor_result), report_content=report_content)
|
||||
system_prompt = T(".prompts:hypothesis_generation.system").r()
|
||||
user_prompt = T(".prompts:hypothesis_generation.user").r(
|
||||
factor_descriptions=json.dumps(factor_result), report_content=report_content
|
||||
)
|
||||
|
||||
response = APIBackend().build_messages_and_create_chat_completion(
|
||||
user_prompt=user_prompt,
|
||||
system_prompt=system_prompt,
|
||||
json_mode=True,
|
||||
json_target_type=Dict[str, str],
|
||||
)
|
||||
|
||||
response_json = json.loads(response)
|
||||
@@ -64,7 +57,7 @@ def generate_hypothesis(factor_result: dict, report_content: str) -> str:
|
||||
)
|
||||
|
||||
|
||||
def extract_hypothesis_and_exp_from_reports(report_file_path: str) -> Tuple[QlibFactorExperiment, Hypothesis]:
|
||||
def extract_hypothesis_and_exp_from_reports(report_file_path: str) -> QlibFactorExperiment | None:
|
||||
"""
|
||||
Extract hypothesis and experiment details from report files.
|
||||
|
||||
@@ -72,17 +65,15 @@ def extract_hypothesis_and_exp_from_reports(report_file_path: str) -> Tuple[Qlib
|
||||
report_file_path (str): Path to the report file.
|
||||
|
||||
Returns:
|
||||
Tuple[QlibFactorExperiment, Hypothesis]: The extracted experiment and generated hypothesis.
|
||||
QlibFactorExperiment: An instance of QlibFactorExperiment containing the extracted details.
|
||||
None: If no valid experiment is found in the report.
|
||||
"""
|
||||
with logger.tag("extract_factors_and_implement"):
|
||||
with logger.tag("load_factor_tasks"):
|
||||
exp = FactorExperimentLoaderFromPDFfiles().load(report_file_path)
|
||||
if exp is None or exp.sub_tasks == []:
|
||||
return None, None
|
||||
exp = FactorExperimentLoaderFromPDFfiles().load(report_file_path)
|
||||
if exp is None or exp.sub_tasks == []:
|
||||
return None
|
||||
|
||||
with logger.tag("load_pdf_screenshot"):
|
||||
pdf_screenshot = extract_first_page_screenshot_from_pdf(report_file_path)
|
||||
logger.log_object(pdf_screenshot)
|
||||
pdf_screenshot = extract_first_page_screenshot_from_pdf(report_file_path)
|
||||
logger.log_object(pdf_screenshot, tag="load_pdf_screenshot")
|
||||
|
||||
docs_dict = load_and_process_pdfs_by_langchain(report_file_path)
|
||||
|
||||
@@ -98,11 +89,11 @@ def extract_hypothesis_and_exp_from_reports(report_file_path: str) -> Tuple[Qlib
|
||||
|
||||
report_content = "\n".join(docs_dict.values())
|
||||
hypothesis = generate_hypothesis(factor_result, report_content)
|
||||
return exp, hypothesis
|
||||
exp.hypothesis = hypothesis
|
||||
return exp
|
||||
|
||||
|
||||
class FactorReportLoop(FactorRDLoop, metaclass=LoopMeta):
|
||||
@measure_time
|
||||
def __init__(self, report_folder: str = None):
|
||||
super().__init__(PROP_SETTING=FACTOR_FROM_REPORT_PROP_SETTING)
|
||||
if report_folder is None:
|
||||
@@ -112,44 +103,31 @@ class FactorReportLoop(FactorRDLoop, metaclass=LoopMeta):
|
||||
else:
|
||||
self.judge_pdf_data_items = [i for i in Path(report_folder).rglob("*.pdf")]
|
||||
|
||||
self.pdf_file_index = 0
|
||||
self.valid_pdf_file_count = 0
|
||||
self.current_loop_hypothesis = None
|
||||
self.current_loop_exp = None
|
||||
self.steps = ["propose_hypo_exp", "propose", "exp_gen", "coding", "running", "feedback"]
|
||||
self.loop_n = min(len(self.judge_pdf_data_items), FACTOR_FROM_REPORT_PROP_SETTING.report_limit)
|
||||
|
||||
@measure_time
|
||||
def propose_hypo_exp(self, prev_out: dict[str, Any]):
|
||||
with logger.tag("r"):
|
||||
while True:
|
||||
if self.valid_pdf_file_count > 15:
|
||||
break
|
||||
report_file_path = self.judge_pdf_data_items[self.pdf_file_index]
|
||||
logger.info(f"Processing number {self.pdf_file_index} report: {report_file_path}")
|
||||
self.pdf_file_index += 1
|
||||
exp, hypothesis = extract_hypothesis_and_exp_from_reports(str(report_file_path))
|
||||
if exp is None:
|
||||
continue
|
||||
self.valid_pdf_file_count += 1
|
||||
exp.based_experiments = [QlibFactorExperiment(sub_tasks=[])] + [t[1] for t in self.trace.hist if t[2]]
|
||||
exp.sub_workspace_list = exp.sub_workspace_list[: FACTOR_FROM_REPORT_PROP_SETTING.max_factors_per_exp]
|
||||
exp.sub_tasks = exp.sub_tasks[: FACTOR_FROM_REPORT_PROP_SETTING.max_factors_per_exp]
|
||||
logger.log_object(hypothesis, tag="hypothesis generation")
|
||||
logger.log_object(exp.sub_tasks, tag="experiment generation")
|
||||
self.current_loop_hypothesis = hypothesis
|
||||
self.current_loop_exp = exp
|
||||
return None
|
||||
def direct_exp_gen(self, prev_out: dict[str, Any]):
|
||||
while True:
|
||||
report_file_path = self.judge_pdf_data_items[self.loop_idx]
|
||||
logger.info(f"Processing number {self.loop_idx} report: {report_file_path}")
|
||||
exp = extract_hypothesis_and_exp_from_reports(str(report_file_path))
|
||||
if exp is None:
|
||||
continue
|
||||
exp.based_experiments = [QlibFactorExperiment(sub_tasks=[], hypothesis=exp.hypothesis)] + [
|
||||
t[0] for t in self.trace.hist if t[1]
|
||||
]
|
||||
exp.sub_workspace_list = exp.sub_workspace_list[: FACTOR_FROM_REPORT_PROP_SETTING.max_factors_per_exp]
|
||||
exp.sub_tasks = exp.sub_tasks[: FACTOR_FROM_REPORT_PROP_SETTING.max_factors_per_exp]
|
||||
logger.log_object(exp.hypothesis, tag="hypothesis generation")
|
||||
logger.log_object(exp.sub_tasks, tag="experiment generation")
|
||||
return exp
|
||||
|
||||
@measure_time
|
||||
def propose(self, prev_out: dict[str, Any]):
|
||||
return self.current_loop_hypothesis
|
||||
|
||||
@measure_time
|
||||
def exp_gen(self, prev_out: dict[str, Any]):
|
||||
return self.current_loop_exp
|
||||
def coding(self, prev_out: dict[str, Any]):
|
||||
exp = self.coder.develop(prev_out["direct_exp_gen"])
|
||||
logger.log_object(exp.sub_workspace_list, tag="coder result")
|
||||
return exp
|
||||
|
||||
|
||||
def main(report_folder=None, path=None, step_n=None):
|
||||
def main(report_folder=None, path=None, all_duration=None, checkout=True):
|
||||
"""
|
||||
Auto R&D Evolving loop for fintech factors (the factors are extracted from finance reports).
|
||||
|
||||
@@ -161,11 +139,11 @@ def main(report_folder=None, path=None, step_n=None):
|
||||
if path is None and report_folder is None:
|
||||
model_loop = FactorReportLoop()
|
||||
elif path is not None:
|
||||
model_loop = FactorReportLoop.load(path)
|
||||
model_loop = FactorReportLoop.load(path, checkout=checkout)
|
||||
else:
|
||||
model_loop = FactorReportLoop(report_folder=report_folder)
|
||||
|
||||
model_loop.run(step_n=step_n)
|
||||
asyncio.run(model_loop.run(all_duration=all_duration))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -2,6 +2,8 @@
|
||||
Model workflow with session control
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
|
||||
import fire
|
||||
|
||||
from rdagent.app.qlib_rd_loop.conf import MODEL_PROP_SETTING
|
||||
@@ -13,7 +15,7 @@ class ModelRDLoop(RDLoop):
|
||||
skip_loop_error = (ModelEmptyError,)
|
||||
|
||||
|
||||
def main(path=None, step_n=None):
|
||||
def main(path=None, step_n=None, loop_n=None, all_duration=None, checkout=True):
|
||||
"""
|
||||
Auto R&D Evolving loop for fintech models
|
||||
|
||||
@@ -27,8 +29,8 @@ def main(path=None, step_n=None):
|
||||
if path is None:
|
||||
model_loop = ModelRDLoop(MODEL_PROP_SETTING)
|
||||
else:
|
||||
model_loop = ModelRDLoop.load(path)
|
||||
model_loop.run(step_n=step_n)
|
||||
model_loop = ModelRDLoop.load(path, checkout=checkout)
|
||||
asyncio.run(model_loop.run(step_n=step_n, loop_n=loop_n, all_duration=all_duration))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -5,10 +5,6 @@ hypothesis_generation:
|
||||
{
|
||||
"hypothesis": "A clear and concise hypothesis based on the provided information.",
|
||||
"reason": "A detailed explanation supporting the generated hypothesis.",
|
||||
"concise_reason": "One line summary that focuses on the justification for the change that leads to the hypothesis (like a part of a knowledge that we are building)",
|
||||
"concise_observation": "One line summary. It focuses on the observation of the given scenario, data characteristics, or previous experiences (failures & succeses).",
|
||||
"concise_justification": "One line summary. It focuses on the justification for the change in new hypothesis and the route of exploration supporting the growth of the hypothesis, based on the observation. ",
|
||||
"concise_knowledge": "One line summary. It focuses on a transferable knowledege that comes with the new hypothesis. Use conditional grammar. eg. "If...., ..; When..., .; and etc"
|
||||
}
|
||||
|
||||
user: |-
|
||||
|
||||
@@ -0,0 +1,133 @@
|
||||
"""
|
||||
Quant (Factor & Model) workflow with session control
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
from typing import Any
|
||||
|
||||
import fire
|
||||
|
||||
from rdagent.app.qlib_rd_loop.conf import QUANT_PROP_SETTING
|
||||
from rdagent.components.workflow.conf import BasePropSetting
|
||||
from rdagent.components.workflow.rd_loop import RDLoop
|
||||
from rdagent.core.developer import Developer
|
||||
from rdagent.core.exception import FactorEmptyError, ModelEmptyError
|
||||
from rdagent.core.proposal import (
|
||||
Experiment2Feedback,
|
||||
Hypothesis2Experiment,
|
||||
HypothesisFeedback,
|
||||
HypothesisGen,
|
||||
)
|
||||
from rdagent.core.scenario import Scenario
|
||||
from rdagent.core.utils import import_class
|
||||
from rdagent.log import rdagent_logger as logger
|
||||
from rdagent.scenarios.qlib.proposal.quant_proposal import QuantTrace
|
||||
|
||||
|
||||
class QuantRDLoop(RDLoop):
|
||||
skip_loop_error = (
|
||||
FactorEmptyError,
|
||||
ModelEmptyError,
|
||||
)
|
||||
|
||||
def __init__(self, PROP_SETTING: BasePropSetting):
|
||||
scen: Scenario = import_class(PROP_SETTING.scen)()
|
||||
logger.log_object(scen, tag="scenario")
|
||||
|
||||
self.hypothesis_gen: HypothesisGen = import_class(PROP_SETTING.quant_hypothesis_gen)(scen)
|
||||
logger.log_object(self.hypothesis_gen, tag="quant hypothesis generator")
|
||||
|
||||
self.factor_hypothesis2experiment: Hypothesis2Experiment = import_class(
|
||||
PROP_SETTING.factor_hypothesis2experiment
|
||||
)()
|
||||
logger.log_object(self.factor_hypothesis2experiment, tag="factor hypothesis2experiment")
|
||||
self.model_hypothesis2experiment: Hypothesis2Experiment = import_class(
|
||||
PROP_SETTING.model_hypothesis2experiment
|
||||
)()
|
||||
logger.log_object(self.model_hypothesis2experiment, tag="model hypothesis2experiment")
|
||||
|
||||
self.factor_coder: Developer = import_class(PROP_SETTING.factor_coder)(scen)
|
||||
logger.log_object(self.factor_coder, tag="factor coder")
|
||||
self.model_coder: Developer = import_class(PROP_SETTING.model_coder)(scen)
|
||||
logger.log_object(self.model_coder, tag="model coder")
|
||||
|
||||
self.factor_runner: Developer = import_class(PROP_SETTING.factor_runner)(scen)
|
||||
logger.log_object(self.factor_runner, tag="factor runner")
|
||||
self.model_runner: Developer = import_class(PROP_SETTING.model_runner)(scen)
|
||||
logger.log_object(self.model_runner, tag="model runner")
|
||||
|
||||
self.factor_summarizer: Experiment2Feedback = import_class(PROP_SETTING.factor_summarizer)(scen)
|
||||
logger.log_object(self.factor_summarizer, tag="factor summarizer")
|
||||
self.model_summarizer: Experiment2Feedback = import_class(PROP_SETTING.model_summarizer)(scen)
|
||||
logger.log_object(self.model_summarizer, tag="model summarizer")
|
||||
|
||||
self.trace = QuantTrace(scen=scen)
|
||||
super(RDLoop, self).__init__()
|
||||
|
||||
def direct_exp_gen(self, prev_out: dict[str, Any]):
|
||||
hypo = self._propose()
|
||||
assert hypo.action in ["factor", "model"]
|
||||
if hypo.action == "factor":
|
||||
exp = self.factor_hypothesis2experiment.convert(hypo, self.trace)
|
||||
else:
|
||||
exp = self.model_hypothesis2experiment.convert(hypo, self.trace)
|
||||
logger.log_object(exp.sub_tasks, tag="experiment generation")
|
||||
return {"propose": hypo, "exp_gen": exp}
|
||||
|
||||
def coding(self, prev_out: dict[str, Any]):
|
||||
if prev_out["direct_exp_gen"]["propose"].action == "factor":
|
||||
exp = self.factor_coder.develop(prev_out["direct_exp_gen"]["exp_gen"])
|
||||
elif prev_out["direct_exp_gen"]["propose"].action == "model":
|
||||
exp = self.model_coder.develop(prev_out["direct_exp_gen"]["exp_gen"])
|
||||
logger.log_object(exp, tag="coder result")
|
||||
return exp
|
||||
|
||||
def running(self, prev_out: dict[str, Any]):
|
||||
if prev_out["direct_exp_gen"]["propose"].action == "factor":
|
||||
exp = self.factor_runner.develop(prev_out["coding"])
|
||||
if exp is None:
|
||||
logger.error(f"Factor extraction failed.")
|
||||
raise FactorEmptyError("Factor extraction failed.")
|
||||
elif prev_out["direct_exp_gen"]["propose"].action == "model":
|
||||
exp = self.model_runner.develop(prev_out["coding"])
|
||||
logger.log_object(exp, tag="runner result")
|
||||
return exp
|
||||
|
||||
def feedback(self, prev_out: dict[str, Any]):
|
||||
e = prev_out.get(self.EXCEPTION_KEY, None)
|
||||
if e is not None:
|
||||
feedback = HypothesisFeedback(
|
||||
observations=str(e),
|
||||
hypothesis_evaluation="",
|
||||
new_hypothesis="",
|
||||
reason="",
|
||||
decision=False,
|
||||
)
|
||||
logger.log_object(feedback, tag="feedback")
|
||||
self.trace.hist.append((prev_out["direct_exp_gen"]["exp_gen"], feedback))
|
||||
else:
|
||||
if prev_out["direct_exp_gen"]["propose"].action == "factor":
|
||||
feedback = self.factor_summarizer.generate_feedback(prev_out["running"], self.trace)
|
||||
elif prev_out["direct_exp_gen"]["propose"].action == "model":
|
||||
feedback = self.model_summarizer.generate_feedback(prev_out["running"], self.trace)
|
||||
logger.log_object(feedback, tag="feedback")
|
||||
self.trace.hist.append((prev_out["running"], feedback))
|
||||
|
||||
|
||||
def main(path=None, step_n=None, loop_n=None, all_duration=None, checkout=True):
|
||||
"""
|
||||
Auto R&D Evolving loop for fintech factors.
|
||||
You can continue running session by
|
||||
.. code-block:: python
|
||||
dotenv run -- python rdagent/app/qlib_rd_loop/quant.py $LOG_PATH/__session__/1/0_propose --step_n 1 # `step_n` is a optional paramter
|
||||
"""
|
||||
if path is None:
|
||||
quant_loop = QuantRDLoop(QUANT_PROP_SETTING)
|
||||
else:
|
||||
quant_loop = QuantRDLoop.load(path, checkout=checkout)
|
||||
|
||||
asyncio.run(quant_loop.run(step_n=step_n, loop_n=loop_n, all_duration=all_duration))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
fire.Fire(main)
|
||||
@@ -0,0 +1,49 @@
|
||||
"""
|
||||
This is the preliminary version of the APE (Automated Prompt Engineering)
|
||||
"""
|
||||
|
||||
import pickle
|
||||
from pathlib import Path
|
||||
|
||||
from rdagent.log.conf import LOG_SETTINGS
|
||||
|
||||
|
||||
def get_llm_qa(file_path):
|
||||
data_flt = []
|
||||
with open(file_path, "rb") as f:
|
||||
data = pickle.load(f)
|
||||
print(len(data))
|
||||
for item in data:
|
||||
if "debug_llm" in item["tag"]:
|
||||
data_flt.append(item)
|
||||
return data_flt
|
||||
|
||||
|
||||
# Example usage
|
||||
# use
|
||||
file_path = Path(LOG_SETTINGS.trace_path) / "debug_llm.pkl"
|
||||
llm_qa = get_llm_qa(file_path)
|
||||
print(len(llm_qa))
|
||||
|
||||
print(llm_qa[0])
|
||||
|
||||
# Initialize APE backend
|
||||
from rdagent.oai.llm_utils import APIBackend
|
||||
from rdagent.utils.agent.tpl import T
|
||||
|
||||
api = APIBackend()
|
||||
|
||||
# Analyze test data and generate improved prompts
|
||||
for qa in llm_qa:
|
||||
# Generate system prompt for APE
|
||||
system_prompt = T(".prompts:ape.system").r()
|
||||
|
||||
# Generate user prompt with context from LLM QA
|
||||
user_prompt = T(".prompts:ape.user").r(
|
||||
system=qa["obj"].get("system", ""), user=qa["obj"]["user"], answer=qa["obj"]["resp"]
|
||||
)
|
||||
analysis_result = api.build_messages_and_create_chat_completion(
|
||||
system_prompt=system_prompt, user_prompt=user_prompt
|
||||
)
|
||||
print(f"█" * 60)
|
||||
yes = input("Do you want to continue? (y/n)")
|
||||
@@ -0,0 +1,49 @@
|
||||
import socket
|
||||
|
||||
import docker
|
||||
|
||||
from rdagent.log import rdagent_logger as logger
|
||||
|
||||
|
||||
def check_docker() -> None:
|
||||
try:
|
||||
client = docker.from_env()
|
||||
client.images.pull("hello-world")
|
||||
container = client.containers.run("hello-world", detach=True)
|
||||
logs = container.logs().decode("utf-8")
|
||||
print(logs)
|
||||
container.remove()
|
||||
logger.info(f"The docker status is normal")
|
||||
except docker.errors.DockerException as e:
|
||||
logger.error(f"An error occurred: {e}")
|
||||
logger.warning(
|
||||
f"Docker status is exception, please check the docker configuration or reinstall it. Refs: https://docs.docker.com/engine/install/ubuntu/."
|
||||
)
|
||||
|
||||
|
||||
def is_port_in_use(port):
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
|
||||
return s.connect_ex(("127.0.0.1", port)) == 0
|
||||
|
||||
|
||||
def check_and_list_free_ports(start_port=19899, max_ports=10) -> None:
|
||||
is_occupied = is_port_in_use(port=start_port)
|
||||
if is_occupied:
|
||||
free_ports = []
|
||||
for port in range(start_port, start_port + max_ports):
|
||||
if not is_port_in_use(port):
|
||||
free_ports.append(port)
|
||||
logger.warning(
|
||||
f"Port 19899 is occupied, please replace it with an available port when running the `rdagent ui` command. Available ports: {free_ports}"
|
||||
)
|
||||
else:
|
||||
logger.info(f"Port 19899 is not occupied, you can run the `rdagent ui` command")
|
||||
|
||||
|
||||
def health_check():
|
||||
"""
|
||||
Check that docker is installed correctly,
|
||||
and that the ports used in the sample README are not occupied.
|
||||
"""
|
||||
check_docker()
|
||||
check_and_list_free_ports()
|
||||
@@ -0,0 +1,119 @@
|
||||
ape:
|
||||
system: |-
|
||||
We'll provide you with a pair of Chat QA about data science.
|
||||
We are creating solutions for a Kaggle Competition based on the answers.
|
||||
Good questions are crucial for getting good answers.
|
||||
Please suggest how to improve the question.
|
||||
You can analyze based on these aspects:
|
||||
- Is the question complete (is all the information needed to answer the question provided?)
|
||||
|
||||
The conversation will be provided in the following format:
|
||||
|
||||
<question>
|
||||
<part1>
|
||||
...text to describe the question...
|
||||
</part1>
|
||||
<part2>
|
||||
...text to describe the question...
|
||||
</part2>
|
||||
</question>
|
||||
|
||||
<answer>
|
||||
...text to describe the answer.
|
||||
</answer>
|
||||
|
||||
You response should be very concorete and concise(less than 20 words) and focuse on the mentioned aspects, like
|
||||
```
|
||||
Info Missing: the question ask for changing code, but it does not provide the description of current code.
|
||||
```
|
||||
Please be very conversatiive when you propose improvements. Only propose improvements when it becomes impossible to give the answer.
|
||||
|
||||
Don't propose conerete modifications
|
||||
|
||||
user: |-
|
||||
<question>
|
||||
<part1>
|
||||
{{system}}
|
||||
</part1>
|
||||
<part2>
|
||||
{{user}}
|
||||
</part2>
|
||||
</question>
|
||||
|
||||
<answer>
|
||||
{{answer}}
|
||||
</answer>
|
||||
|
||||
optional: |-
|
||||
If you want to suggest modification on the question. Please follow the *SEARCH/REPLACE block* Rules!!!! It is optional.
|
||||
Please make it concise and less than 20 lines!!!
|
||||
|
||||
# *SEARCH/REPLACE block* Rules:
|
||||
|
||||
Every *SEARCH/REPLACE block* must use this format:
|
||||
1. The *FULL* file path alone on a line, verbatim. No bold asterisks, no quotes around it, no escaping of characters, etc.
|
||||
2. The opening fence and code language, eg: ```python
|
||||
3. The start of search block: <<<<<<< SEARCH
|
||||
4. A contiguous chunk of lines to search for in the existing source code
|
||||
5. The dividing line: =======
|
||||
6. The lines to replace into the source code
|
||||
7. The end of the replace block: >>>>>>> REPLACE
|
||||
8. The closing fence: ```
|
||||
|
||||
Use the *FULL* file path, as shown to you by the user.
|
||||
|
||||
Every *SEARCH* section must *EXACTLY MATCH* the existing file content, character for character, including all comments, docstrings, etc.
|
||||
If the file contains code or other data wrapped/escaped in json/xml/quotes or other containers, you need to propose edits to the literal contents of the file, including the container markup.
|
||||
|
||||
*SEARCH/REPLACE* blocks will *only* replace the first match occurrence.
|
||||
Including multiple unique *SEARCH/REPLACE* blocks if needed.
|
||||
Include enough lines in each SEARCH section to uniquely match each set of lines that need to change.
|
||||
|
||||
Keep *SEARCH/REPLACE* blocks concise.
|
||||
Break large *SEARCH/REPLACE* blocks into a series of smaller blocks that each change a small portion of the file.
|
||||
Include just the changing lines, and a few surrounding lines if needed for uniqueness.
|
||||
Do not include long runs of unchanging lines in *SEARCH/REPLACE* blocks.
|
||||
|
||||
Only create *SEARCH/REPLACE* blocks for files that the user has added to the chat!
|
||||
|
||||
To move code within a file, use 2 *SEARCH/REPLACE* blocks: 1 to delete it from its current location, 1 to insert it in the new location.
|
||||
|
||||
Pay attention to which filenames the user wants you to edit, especially if they are asking you to create a new file.
|
||||
|
||||
If you want to put code in a new file, use a *SEARCH/REPLACE block* with:
|
||||
- A new file path, including dir name if needed
|
||||
- An empty `SEARCH` section
|
||||
- The new file's contents in the `REPLACE` section
|
||||
|
||||
To rename files which have been added to the chat, use shell commands at the end of your response.
|
||||
|
||||
If the user just says something like "ok" or "go ahead" or "do that" they probably want you to make SEARCH/REPLACE blocks for the code changes you just proposed.
|
||||
The user will say when they've applied your edits. If they haven't explicitly confirmed the edits have been applied, they probably want proper SEARCH/REPLACE blocks.
|
||||
|
||||
You are diligent and tireless!
|
||||
You NEVER leave comments describing code without implementing it!
|
||||
You always COMPLETELY IMPLEMENT the needed code!
|
||||
|
||||
|
||||
ONLY EVER RETURN CODE IN A *SEARCH/REPLACE BLOCK*!
|
||||
Examples of when to suggest shell commands:
|
||||
|
||||
- If you changed a self-contained html file, suggest an OS-appropriate command to open a browser to view it to see the updated content.
|
||||
- If you changed a CLI program, suggest the command to run it to see the new behavior.
|
||||
- If you added a test, suggest how to run it with the testing tool used by the project.
|
||||
- Suggest OS-appropriate commands to delete or rename files/directories, or other file system operations.
|
||||
- If your code changes add new dependencies, suggest the command to install them.
|
||||
- Etc.
|
||||
|
||||
Here is a example of SEARCH/REPLACE BLOCK to change a function implementation to import.
|
||||
|
||||
<<<<<<< SEARCH
|
||||
def hello():
|
||||
"print a greeting"
|
||||
|
||||
print("hello")
|
||||
=======
|
||||
from hello import hello
|
||||
|
||||
>>>>>>> REPLACE
|
||||
# - Is there any ambiguity in the question?
|
||||
@@ -2,19 +2,16 @@ from dataclasses import field
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from pydantic_settings import BaseSettings
|
||||
from rdagent.core.conf import ExtendedBaseSettings
|
||||
|
||||
DIRNAME = Path("./")
|
||||
|
||||
|
||||
class BenchmarkSettings(BaseSettings):
|
||||
class BenchmarkSettings(ExtendedBaseSettings):
|
||||
class Config:
|
||||
env_prefix = "BENCHMARK_"
|
||||
"""Use `BENCHMARK_` as prefix for environment variables"""
|
||||
|
||||
ground_truth_dir: Path = DIRNAME / "ground_truth"
|
||||
"""ground truth dir"""
|
||||
|
||||
bench_data_path: Path = DIRNAME / "example.json"
|
||||
"""data for benchmark"""
|
||||
|
||||
@@ -24,7 +21,7 @@ class BenchmarkSettings(BaseSettings):
|
||||
bench_test_case_n: Optional[int] = None
|
||||
"""how many test cases to run; If not given, all test cases will be run"""
|
||||
|
||||
bench_method_cls: str = "rdagent.components.coder.factor_coder.CoSTEER.FactorCoSTEER"
|
||||
bench_method_cls: str = "rdagent.components.coder.factor_coder.FactorCoSTEER"
|
||||
"""method to be used for test cases"""
|
||||
|
||||
bench_method_extra_kwargs: dict = field(
|
||||
|
||||
@@ -5,8 +5,8 @@ from typing import Dict, List, Tuple, Union
|
||||
import pandas as pd
|
||||
from tqdm import tqdm
|
||||
|
||||
from rdagent.components.coder.factor_coder.config import FACTOR_IMPLEMENT_SETTINGS
|
||||
from rdagent.components.coder.factor_coder.CoSTEER.evaluators import (
|
||||
from rdagent.components.coder.factor_coder.config import FACTOR_COSTEER_SETTINGS
|
||||
from rdagent.components.coder.factor_coder.eva_utils import (
|
||||
FactorCorrelationEvaluator,
|
||||
FactorEqualValueRatioEvaluator,
|
||||
FactorEvaluator,
|
||||
|
||||
@@ -0,0 +1,144 @@
|
||||
import pickle
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
|
||||
from rdagent.components.coder.CoSTEER.config import CoSTEERSettings
|
||||
from rdagent.components.coder.CoSTEER.evaluators import CoSTEERMultiFeedback
|
||||
from rdagent.components.coder.CoSTEER.evolvable_subjects import EvolvingItem
|
||||
from rdagent.components.coder.CoSTEER.knowledge_management import (
|
||||
CoSTEERKnowledgeBaseV1,
|
||||
CoSTEERKnowledgeBaseV2,
|
||||
CoSTEERRAGStrategyV1,
|
||||
CoSTEERRAGStrategyV2,
|
||||
)
|
||||
from rdagent.core.developer import Developer
|
||||
from rdagent.core.evaluation import Evaluator, Feedback
|
||||
from rdagent.core.evolving_agent import EvolvingStrategy, RAGEvoAgent
|
||||
from rdagent.core.exception import CoderError
|
||||
from rdagent.core.experiment import Experiment
|
||||
from rdagent.log import rdagent_logger as logger
|
||||
|
||||
|
||||
class CoSTEER(Developer[Experiment]):
|
||||
def __init__(
|
||||
self,
|
||||
settings: CoSTEERSettings,
|
||||
eva: Evaluator,
|
||||
es: EvolvingStrategy,
|
||||
evolving_version: int,
|
||||
*args,
|
||||
with_knowledge: bool = True,
|
||||
with_feedback: bool = True,
|
||||
knowledge_self_gen: bool = True,
|
||||
filter_final_evo: bool = True,
|
||||
max_loop: int | None = None,
|
||||
**kwargs,
|
||||
) -> None:
|
||||
super().__init__(*args, **kwargs)
|
||||
self.max_loop = settings.max_loop if max_loop is None else max_loop
|
||||
self.max_seconds = settings.max_seconds
|
||||
self.knowledge_base_path = (
|
||||
Path(settings.knowledge_base_path) if settings.knowledge_base_path is not None else None
|
||||
)
|
||||
self.new_knowledge_base_path = (
|
||||
Path(settings.new_knowledge_base_path) if settings.new_knowledge_base_path is not None else None
|
||||
)
|
||||
|
||||
self.with_knowledge = with_knowledge
|
||||
self.with_feedback = with_feedback
|
||||
self.knowledge_self_gen = knowledge_self_gen
|
||||
self.filter_final_evo = filter_final_evo
|
||||
self.evolving_strategy = es
|
||||
self.evaluator = eva
|
||||
self.evolving_version = evolving_version
|
||||
|
||||
# init knowledge base
|
||||
self.knowledge_base = self.load_or_init_knowledge_base(
|
||||
former_knowledge_base_path=self.knowledge_base_path,
|
||||
component_init_list=[],
|
||||
)
|
||||
# init rag method
|
||||
self.rag = (
|
||||
CoSTEERRAGStrategyV2(self.knowledge_base, settings=settings)
|
||||
if self.evolving_version == 2
|
||||
else CoSTEERRAGStrategyV1(self.knowledge_base, settings=settings)
|
||||
)
|
||||
|
||||
def load_or_init_knowledge_base(self, former_knowledge_base_path: Path = None, component_init_list: list = []):
|
||||
if former_knowledge_base_path is not None and former_knowledge_base_path.exists():
|
||||
knowledge_base = pickle.load(open(former_knowledge_base_path, "rb"))
|
||||
if self.evolving_version == 1 and not isinstance(knowledge_base, CoSTEERKnowledgeBaseV1):
|
||||
raise ValueError("The former knowledge base is not compatible with the current version")
|
||||
elif self.evolving_version == 2 and not isinstance(
|
||||
knowledge_base,
|
||||
CoSTEERKnowledgeBaseV2,
|
||||
):
|
||||
raise ValueError("The former knowledge base is not compatible with the current version")
|
||||
else:
|
||||
knowledge_base = (
|
||||
CoSTEERKnowledgeBaseV2(
|
||||
init_component_list=component_init_list,
|
||||
)
|
||||
if self.evolving_version == 2
|
||||
else CoSTEERKnowledgeBaseV1()
|
||||
)
|
||||
return knowledge_base
|
||||
|
||||
def develop(self, exp: Experiment) -> Experiment:
|
||||
|
||||
# init intermediate items
|
||||
evo_exp = EvolvingItem.from_experiment(exp)
|
||||
|
||||
self.evolve_agent = RAGEvoAgent(
|
||||
max_loop=self.max_loop,
|
||||
evolving_strategy=self.evolving_strategy,
|
||||
rag=self.rag,
|
||||
with_knowledge=self.with_knowledge,
|
||||
with_feedback=self.with_feedback,
|
||||
knowledge_self_gen=self.knowledge_self_gen,
|
||||
)
|
||||
|
||||
start_datetime = datetime.now()
|
||||
for evo_exp in self.evolve_agent.multistep_evolve(evo_exp, self.evaluator):
|
||||
assert isinstance(evo_exp, Experiment) # multiple inheritance
|
||||
logger.log_object(evo_exp.sub_workspace_list, tag="evolving code")
|
||||
for sw in evo_exp.sub_workspace_list:
|
||||
logger.info(f"evolving workspace: {sw}")
|
||||
if (datetime.now() - start_datetime).seconds > self.max_seconds:
|
||||
logger.info(f"Reached max time limit {self.max_seconds} seconds, stop evolving")
|
||||
break
|
||||
|
||||
if self.with_feedback and self.filter_final_evo:
|
||||
evo_exp = self._exp_postprocess_by_feedback(evo_exp, self.evolve_agent.evolving_trace[-1].feedback)
|
||||
|
||||
# save new knowledge base
|
||||
if self.new_knowledge_base_path is not None:
|
||||
with self.new_knowledge_base_path.open("wb") as f:
|
||||
pickle.dump(self.knowledge_base, f)
|
||||
logger.info(f"New knowledge base saved to {self.new_knowledge_base_path}")
|
||||
exp.sub_workspace_list = evo_exp.sub_workspace_list
|
||||
exp.experiment_workspace = evo_exp.experiment_workspace
|
||||
return exp
|
||||
|
||||
def _exp_postprocess_by_feedback(self, evo: Experiment, feedback: CoSTEERMultiFeedback) -> Experiment:
|
||||
"""
|
||||
Responsibility:
|
||||
- Raise Error if it failed to handle the develop task
|
||||
-
|
||||
"""
|
||||
assert isinstance(evo, Experiment)
|
||||
assert isinstance(feedback, CoSTEERMultiFeedback)
|
||||
assert len(evo.sub_workspace_list) == len(feedback)
|
||||
|
||||
# FIXME: when whould the feedback be None?
|
||||
failed_feedbacks = [
|
||||
f"- feedback{index + 1:02d}:\n - execution: {f.execution}\n - return_checking: {f.return_checking}\n - code: {f.code}"
|
||||
for index, f in enumerate(feedback)
|
||||
if f is not None and not f.final_decision
|
||||
]
|
||||
|
||||
if len(failed_feedbacks) == len(feedback):
|
||||
feedback_summary = "\n".join(failed_feedbacks)
|
||||
raise CoderError(f"All tasks are failed:\n{feedback_summary}")
|
||||
|
||||
return evo
|
||||
@@ -0,0 +1,39 @@
|
||||
from typing import Union
|
||||
|
||||
from rdagent.core.conf import ExtendedBaseSettings
|
||||
|
||||
|
||||
class CoSTEERSettings(ExtendedBaseSettings):
|
||||
"""CoSTEER settings, this setting is supposed not to be used directly!!!"""
|
||||
|
||||
class Config:
|
||||
env_prefix = "CoSTEER_"
|
||||
|
||||
coder_use_cache: bool = False
|
||||
"""Indicates whether to use cache for the coder"""
|
||||
|
||||
max_loop: int = 10
|
||||
"""Maximum number of task implementation loops"""
|
||||
|
||||
fail_task_trial_limit: int = 20
|
||||
|
||||
v1_query_former_trace_limit: int = 5
|
||||
v1_query_similar_success_limit: int = 5
|
||||
|
||||
v2_query_component_limit: int = 1
|
||||
v2_query_error_limit: int = 1
|
||||
v2_query_former_trace_limit: int = 1
|
||||
v2_add_fail_attempt_to_latest_successful_execution: bool = False
|
||||
v2_error_summary: bool = False
|
||||
v2_knowledge_sampler: float = 1.0
|
||||
|
||||
knowledge_base_path: Union[str, None] = None
|
||||
"""Path to the knowledge base"""
|
||||
|
||||
new_knowledge_base_path: Union[str, None] = None
|
||||
"""Path to the new knowledge base"""
|
||||
|
||||
max_seconds: int = 10**6
|
||||
|
||||
|
||||
CoSTEER_SETTINGS = CoSTEERSettings()
|
||||
@@ -0,0 +1,277 @@
|
||||
from abc import abstractmethod
|
||||
from copy import deepcopy
|
||||
from dataclasses import dataclass
|
||||
from typing import TYPE_CHECKING, List
|
||||
|
||||
from rdagent.components.coder.CoSTEER.evolvable_subjects import EvolvingItem
|
||||
from rdagent.core.conf import RD_AGENT_SETTINGS
|
||||
from rdagent.core.evaluation import Evaluator, Feedback
|
||||
from rdagent.core.evolving_framework import QueriedKnowledge
|
||||
from rdagent.core.experiment import Task, Workspace
|
||||
from rdagent.core.utils import multiprocessing_wrapper
|
||||
from rdagent.log import rdagent_logger as logger
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from rdagent.core.scenario import Scenario
|
||||
|
||||
# TODO:
|
||||
# 1. It seems logically sound, but we currently lack a scenario to apply it.
|
||||
# 2. If it proves to be useful, relocate it to a more general location.
|
||||
#
|
||||
# class FBWorkspaceExeFeedback(Feedback):
|
||||
# """
|
||||
# It pairs with FBWorkspace in the abstract level.
|
||||
# """
|
||||
# # ws: FBWorkspace # potential
|
||||
# stdout: str
|
||||
|
||||
|
||||
@dataclass
|
||||
class CoSTEERSingleFeedback(Feedback):
|
||||
# TODO: (xiao)
|
||||
# it should be more general class for FBWorkspaceExeFeedback
|
||||
# A better name of it may be NormalFeedback
|
||||
# TODO: It should be a general feeddback for CoSTEERR
|
||||
"""
|
||||
The feedback for the data loader evaluation.
|
||||
It is design align the phases of the implemented code
|
||||
- Execution -> Return Value -> Code -> Final Decision
|
||||
"""
|
||||
execution: str
|
||||
# execution_feedback
|
||||
return_checking: str | None # including every check in the testing (constraints about the generated value)
|
||||
# value_feedback, shape_feedback, value_generated_flag
|
||||
code: str
|
||||
final_decision: bool
|
||||
|
||||
@staticmethod
|
||||
def val_and_update_init_dict(data: dict) -> dict:
|
||||
# TODO: (bowen) use a more general method to validate and update the data dictionary before init, like pydantic
|
||||
"""
|
||||
Validates and converts the 'final_decision' field in the given data dictionary.
|
||||
|
||||
Args:
|
||||
data (dict): The data dictionary containing the 'final_decision' field.
|
||||
|
||||
Returns:
|
||||
dict: The updated data dictionary with 'final_decision' as a boolean.
|
||||
|
||||
Raises:
|
||||
ValueError: If 'final_decision' is not present or not a boolean.
|
||||
"""
|
||||
if "final_decision" not in data:
|
||||
raise ValueError("'final_decision' is required")
|
||||
|
||||
if isinstance(data["final_decision"], str):
|
||||
if data["final_decision"] == "false" or data["final_decision"] == "False":
|
||||
data["final_decision"] = False
|
||||
elif data["final_decision"] == "true" or data["final_decision"] == "True":
|
||||
data["final_decision"] = True
|
||||
|
||||
if not isinstance(data["final_decision"], bool):
|
||||
raise ValueError(f"'final_decision' must be a boolean, not {type(data['final_decision'])}")
|
||||
return data
|
||||
|
||||
def __str__(self) -> str:
|
||||
return f"""------------------Execution------------------
|
||||
{self.execution}
|
||||
------------------Return Checking------------------
|
||||
{self.return_checking if self.return_checking is not None else 'No return checking'}
|
||||
------------------Code------------------
|
||||
{self.code}
|
||||
------------------Final Decision------------------
|
||||
This implementation is {'SUCCESS' if self.final_decision else 'FAIL'}.
|
||||
"""
|
||||
|
||||
def __bool__(self):
|
||||
return self.final_decision
|
||||
|
||||
|
||||
class CoSTEERSingleFeedbackDeprecated(CoSTEERSingleFeedback):
|
||||
"""This class is a base class for all code generator feedback to single implementation"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
execution_feedback: str = None,
|
||||
shape_feedback: str = None,
|
||||
code_feedback: str = None,
|
||||
value_feedback: str = None,
|
||||
final_decision: bool = None,
|
||||
final_feedback: str = None,
|
||||
value_generated_flag: bool = None,
|
||||
final_decision_based_on_gt: bool = None,
|
||||
) -> None:
|
||||
self.execution_feedback = execution_feedback
|
||||
self.code_feedback = code_feedback
|
||||
self.value_feedback = value_feedback
|
||||
self.final_decision = final_decision
|
||||
self.final_feedback = final_feedback
|
||||
self.value_generated_flag = value_generated_flag
|
||||
self.final_decision_based_on_gt = final_decision_based_on_gt
|
||||
|
||||
# TODO:
|
||||
# Not general enough. So we should not put them in the general costeer feedback
|
||||
# Instead, we should create subclass for it.
|
||||
self.shape_feedback = shape_feedback # Not general enough. So
|
||||
|
||||
@property
|
||||
def execution(self):
|
||||
return self.execution_feedback
|
||||
|
||||
@execution.setter
|
||||
def execution(self, value):
|
||||
self.execution_feedback = value
|
||||
|
||||
@property
|
||||
def return_checking(self):
|
||||
if self.value_generated_flag:
|
||||
return f"value feedback: {self.value_feedback}\n\nshape feedback: {self.shape_feedback}"
|
||||
return None
|
||||
|
||||
@return_checking.setter
|
||||
def return_checking(self, value):
|
||||
# Since return_checking is derived from value_feedback and shape_feedback,
|
||||
# we don't need to do anything here
|
||||
self.value_feedback = value
|
||||
self.shape_feedback = value
|
||||
|
||||
@property
|
||||
def code(self):
|
||||
return self.code_feedback
|
||||
|
||||
@code.setter
|
||||
def code(self, value):
|
||||
self.code_feedback = value
|
||||
|
||||
def __str__(self) -> str:
|
||||
return f"""------------------Execution Feedback------------------
|
||||
{self.execution_feedback if self.execution_feedback is not None else 'No execution feedback'}
|
||||
------------------Shape Feedback------------------
|
||||
{self.shape_feedback if self.shape_feedback is not None else 'No shape feedback'}
|
||||
------------------Code Feedback------------------
|
||||
{self.code_feedback if self.code_feedback is not None else 'No code feedback'}
|
||||
------------------Value Feedback------------------
|
||||
{self.value_feedback if self.value_feedback is not None else 'No value feedback'}
|
||||
------------------Final Feedback------------------
|
||||
{self.final_feedback if self.final_feedback is not None else 'No final feedback'}
|
||||
------------------Final Decision------------------
|
||||
This implementation is {'SUCCESS' if self.final_decision else 'FAIL'}.
|
||||
"""
|
||||
|
||||
|
||||
class CoSTEERMultiFeedback(Feedback):
|
||||
"""Feedback contains a list, each element is the corresponding feedback for each factor implementation."""
|
||||
|
||||
def __init__(self, feedback_list: List[CoSTEERSingleFeedback]) -> None:
|
||||
self.feedback_list = feedback_list
|
||||
|
||||
def __getitem__(self, index: int) -> CoSTEERSingleFeedback:
|
||||
return self.feedback_list[index]
|
||||
|
||||
def __len__(self) -> int:
|
||||
return len(self.feedback_list)
|
||||
|
||||
def append(self, feedback: CoSTEERSingleFeedback) -> None:
|
||||
self.feedback_list.append(feedback)
|
||||
|
||||
def __iter__(self):
|
||||
return iter(self.feedback_list)
|
||||
|
||||
def finished(self) -> bool:
|
||||
"""
|
||||
In some implementations, tasks may fail multiple times, leading agents to skip the implementation.
|
||||
This results in None feedback. However, we want to accept the correct parts and ignore None feedback.
|
||||
"""
|
||||
return all(feedback.final_decision for feedback in self.feedback_list if feedback is not None)
|
||||
|
||||
def __bool__(self) -> bool:
|
||||
return all(feedback.final_decision for feedback in self.feedback_list)
|
||||
|
||||
|
||||
class CoSTEEREvaluator(Evaluator):
|
||||
def __init__(
|
||||
self,
|
||||
scen: "Scenario",
|
||||
) -> None:
|
||||
self.scen = scen
|
||||
|
||||
# TODO:
|
||||
# I think we should have unified interface for all evaluates, for examples.
|
||||
# So we should adjust the interface of other factors
|
||||
@abstractmethod
|
||||
def evaluate(
|
||||
self,
|
||||
target_task: Task,
|
||||
implementation: Workspace,
|
||||
gt_implementation: Workspace,
|
||||
**kwargs,
|
||||
) -> CoSTEERSingleFeedback:
|
||||
raise NotImplementedError("Please implement the `evaluator` method")
|
||||
|
||||
|
||||
class CoSTEERMultiEvaluator(CoSTEEREvaluator):
|
||||
"""This is for evaluation of experiment. Due to we have multiple tasks, so we will return a list of evaluation feebacks"""
|
||||
|
||||
def __init__(self, single_evaluator: CoSTEEREvaluator | list[CoSTEEREvaluator], *args, **kwargs) -> None:
|
||||
super().__init__(*args, **kwargs)
|
||||
self.single_evaluator = single_evaluator
|
||||
|
||||
def evaluate(
|
||||
self,
|
||||
evo: EvolvingItem,
|
||||
queried_knowledge: QueriedKnowledge = None,
|
||||
**kwargs,
|
||||
) -> CoSTEERMultiFeedback:
|
||||
eval_l = self.single_evaluator if isinstance(self.single_evaluator, list) else [self.single_evaluator]
|
||||
task_li_feedback_li = []
|
||||
for ev in eval_l:
|
||||
multi_implementation_feedback = multiprocessing_wrapper(
|
||||
[
|
||||
(
|
||||
ev.evaluate,
|
||||
(
|
||||
evo.sub_tasks[index],
|
||||
evo.sub_workspace_list[index],
|
||||
evo.sub_gt_implementations[index] if evo.sub_gt_implementations is not None else None,
|
||||
queried_knowledge,
|
||||
),
|
||||
)
|
||||
for index in range(len(evo.sub_tasks))
|
||||
],
|
||||
n=RD_AGENT_SETTINGS.multi_proc_n,
|
||||
)
|
||||
task_li_feedback_li.append(multi_implementation_feedback)
|
||||
# merge the feedbacks
|
||||
merged_task_feedback = []
|
||||
for task_id, fb in enumerate(task_li_feedback_li[0]):
|
||||
fb = deepcopy(fb) # deep copy to make it more robust
|
||||
|
||||
fb.final_decision = all(
|
||||
task_li_feedback[task_id].final_decision for task_li_feedback in task_li_feedback_li
|
||||
)
|
||||
for attr in "execution", "return_checking", "code":
|
||||
setattr(
|
||||
fb,
|
||||
attr,
|
||||
"\n\n".join(
|
||||
[
|
||||
getattr(task_li_feedback[task_id], attr)
|
||||
for task_li_feedback in task_li_feedback_li
|
||||
if getattr(task_li_feedback[task_id], attr) is not None
|
||||
]
|
||||
),
|
||||
)
|
||||
merged_task_feedback.append(fb)
|
||||
|
||||
final_decision = [
|
||||
None if single_feedback is None else single_feedback.final_decision
|
||||
for single_feedback in merged_task_feedback
|
||||
]
|
||||
logger.info(f"Final decisions: {final_decision} True count: {final_decision.count(True)}")
|
||||
|
||||
# TODO: this is to be compatible with factor_implementation;
|
||||
for index in range(len(evo.sub_tasks)):
|
||||
if final_decision[index]:
|
||||
evo.sub_tasks[index].factor_implementation = True
|
||||
|
||||
return CoSTEERMultiFeedback(merged_task_feedback)
|
||||
+6
-11
@@ -1,24 +1,19 @@
|
||||
from rdagent.components.coder.factor_coder.factor import (
|
||||
FactorExperiment,
|
||||
FactorFBWorkspace,
|
||||
FactorTask,
|
||||
)
|
||||
from rdagent.core.evolving_framework import EvolvableSubjects
|
||||
from rdagent.core.experiment import Experiment, FBWorkspace, Task
|
||||
from rdagent.log import rdagent_logger as logger
|
||||
|
||||
|
||||
class FactorEvolvingItem(FactorExperiment, EvolvableSubjects):
|
||||
class EvolvingItem(Experiment, EvolvableSubjects):
|
||||
"""
|
||||
Intermediate item of factor implementation.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
sub_tasks: list[FactorTask],
|
||||
sub_gt_implementations: list[FactorFBWorkspace] = None,
|
||||
sub_tasks: list[Task],
|
||||
sub_gt_implementations: list[FBWorkspace] = None,
|
||||
):
|
||||
FactorExperiment.__init__(self, sub_tasks=sub_tasks)
|
||||
self.corresponding_selection: list = None
|
||||
Experiment.__init__(self, sub_tasks=sub_tasks)
|
||||
if sub_gt_implementations is not None and len(
|
||||
sub_gt_implementations,
|
||||
) != len(self.sub_tasks):
|
||||
@@ -30,7 +25,7 @@ class FactorEvolvingItem(FactorExperiment, EvolvableSubjects):
|
||||
self.sub_gt_implementations = sub_gt_implementations
|
||||
|
||||
@classmethod
|
||||
def from_experiment(cls, exp: FactorExperiment) -> "FactorExperiment":
|
||||
def from_experiment(cls, exp: Experiment) -> Experiment:
|
||||
ei = cls(sub_tasks=exp.sub_tasks)
|
||||
ei.based_experiments = exp.based_experiments
|
||||
ei.experiment_workspace = exp.experiment_workspace
|
||||
@@ -0,0 +1,120 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from abc import abstractmethod
|
||||
|
||||
from rdagent.components.coder.CoSTEER.config import CoSTEERSettings
|
||||
from rdagent.components.coder.CoSTEER.evaluators import (
|
||||
CoSTEERMultiFeedback,
|
||||
CoSTEERSingleFeedback,
|
||||
)
|
||||
from rdagent.components.coder.CoSTEER.evolvable_subjects import EvolvingItem
|
||||
from rdagent.components.coder.CoSTEER.knowledge_management import (
|
||||
CoSTEERQueriedKnowledge,
|
||||
)
|
||||
from rdagent.core.conf import RD_AGENT_SETTINGS
|
||||
from rdagent.core.evolving_framework import EvolvingStrategy, EvoStep, QueriedKnowledge
|
||||
from rdagent.core.experiment import FBWorkspace, Task
|
||||
from rdagent.core.scenario import Scenario
|
||||
from rdagent.core.utils import multiprocessing_wrapper
|
||||
|
||||
|
||||
class MultiProcessEvolvingStrategy(EvolvingStrategy):
|
||||
def __init__(self, scen: Scenario, settings: CoSTEERSettings):
|
||||
super().__init__(scen)
|
||||
self.settings = settings
|
||||
|
||||
@abstractmethod
|
||||
def implement_one_task(
|
||||
self,
|
||||
target_task: Task,
|
||||
queried_knowledge: QueriedKnowledge | None = None,
|
||||
workspace: FBWorkspace | None = None,
|
||||
prev_task_feedback: CoSTEERSingleFeedback | None = None,
|
||||
) -> dict[str, str]: # FIXME: fix interface of previous implement
|
||||
"""
|
||||
This method will input the task & current workspace,
|
||||
and output the modification to applied to the workspace.
|
||||
(i.e. replace the content <filename> with <content>)
|
||||
|
||||
Parameters
|
||||
----------
|
||||
target_task : Task
|
||||
|
||||
queried_knowledge : QueriedKnowledge | None
|
||||
|
||||
workspace : FBWorkspace | None
|
||||
|
||||
prev_task_feedback : CoSTEERSingleFeedback | None
|
||||
task feedback for previous evolving step
|
||||
None indicate it is the first loop.
|
||||
|
||||
Return
|
||||
------
|
||||
The new files {<filename>: <content>} to update the workspace.
|
||||
"""
|
||||
raise NotImplementedError
|
||||
|
||||
@abstractmethod
|
||||
def assign_code_list_to_evo(self, code_list: list[dict], evo: EvolvingItem) -> None:
|
||||
"""
|
||||
Assign the code list to the evolving item.
|
||||
|
||||
Due to the implement_one_task take `workspace` as input and output the `modification`.
|
||||
We should apply implementation to evo
|
||||
|
||||
The code list is aligned with the evolving item's sub-tasks.
|
||||
If a task is not implemented, put a None in the list.
|
||||
"""
|
||||
raise NotImplementedError
|
||||
|
||||
def evolve(
|
||||
self,
|
||||
*,
|
||||
evo: EvolvingItem,
|
||||
queried_knowledge: CoSTEERQueriedKnowledge | None = None,
|
||||
evolving_trace: list[EvoStep] = [],
|
||||
**kwargs,
|
||||
) -> EvolvingItem:
|
||||
# 1.找出需要evolve的task
|
||||
to_be_finished_task_index: list[int] = []
|
||||
for index, target_task in enumerate(evo.sub_tasks):
|
||||
target_task_desc = target_task.get_task_information()
|
||||
if target_task_desc in queried_knowledge.success_task_to_knowledge_dict:
|
||||
# NOTE: very weird logic:
|
||||
# it depends on the knowledge to set the already finished task
|
||||
evo.sub_workspace_list[index] = queried_knowledge.success_task_to_knowledge_dict[
|
||||
target_task_desc
|
||||
].implementation
|
||||
elif (
|
||||
target_task_desc not in queried_knowledge.success_task_to_knowledge_dict
|
||||
and target_task_desc not in queried_knowledge.failed_task_info_set
|
||||
):
|
||||
to_be_finished_task_index.append(index)
|
||||
|
||||
last_feedback = None
|
||||
if len(evolving_trace) > 0:
|
||||
last_feedback = evolving_trace[-1].feedback
|
||||
assert isinstance(last_feedback, CoSTEERMultiFeedback)
|
||||
|
||||
result = multiprocessing_wrapper(
|
||||
[
|
||||
(
|
||||
self.implement_one_task,
|
||||
(
|
||||
evo.sub_tasks[target_index],
|
||||
queried_knowledge,
|
||||
evo.experiment_workspace,
|
||||
None if last_feedback is None else last_feedback[target_index],
|
||||
),
|
||||
)
|
||||
for target_index in to_be_finished_task_index
|
||||
],
|
||||
n=RD_AGENT_SETTINGS.multi_proc_n,
|
||||
)
|
||||
code_list = [None for _ in range(len(evo.sub_tasks))]
|
||||
for index, target_index in enumerate(to_be_finished_task_index):
|
||||
code_list[target_index] = result[index]
|
||||
|
||||
evo = self.assign_code_list_to_evo(code_list, evo)
|
||||
|
||||
return evo
|
||||
+175
-199
@@ -6,19 +6,15 @@ import random
|
||||
import re
|
||||
from itertools import combinations
|
||||
from pathlib import Path
|
||||
from typing import Union
|
||||
from typing import List, Union
|
||||
|
||||
from jinja2 import Environment, StrictUndefined
|
||||
|
||||
from rdagent.components.coder.factor_coder.config import FACTOR_IMPLEMENT_SETTINGS
|
||||
from rdagent.components.coder.factor_coder.CoSTEER.evaluators import (
|
||||
FactorSingleFeedback,
|
||||
)
|
||||
from rdagent.components.coder.factor_coder.factor import FactorTask
|
||||
from rdagent.components.coder.CoSTEER.config import CoSTEERSettings
|
||||
from rdagent.components.coder.CoSTEER.evaluators import CoSTEERSingleFeedback
|
||||
from rdagent.components.knowledge_management.graph import (
|
||||
UndirectedGraph,
|
||||
UndirectedNode,
|
||||
)
|
||||
from rdagent.core.evolving_agent import Feedback
|
||||
from rdagent.core.evolving_framework import (
|
||||
EvolvableSubjects,
|
||||
EvolvingKnowledgeBase,
|
||||
@@ -27,75 +23,73 @@ from rdagent.core.evolving_framework import (
|
||||
QueriedKnowledge,
|
||||
RAGStrategy,
|
||||
)
|
||||
from rdagent.core.experiment import Workspace
|
||||
from rdagent.core.prompts import Prompts
|
||||
from rdagent.core.experiment import FBWorkspace, Task
|
||||
from rdagent.log import rdagent_logger as logger
|
||||
from rdagent.oai.llm_utils import (
|
||||
APIBackend,
|
||||
calculate_embedding_distance_between_str_list,
|
||||
)
|
||||
from rdagent.utils.agent.tpl import T
|
||||
|
||||
|
||||
class FactorKnowledge(Knowledge):
|
||||
class CoSTEERKnowledge(Knowledge):
|
||||
def __init__(
|
||||
self,
|
||||
target_task: FactorTask,
|
||||
implementation: Workspace,
|
||||
feedback: FactorSingleFeedback,
|
||||
target_task: Task,
|
||||
implementation: FBWorkspace,
|
||||
feedback: Feedback,
|
||||
) -> None:
|
||||
"""
|
||||
Initialize a FactorKnowledge object. The FactorKnowledge object is used to store a factor implementation without the ground truth code and value.
|
||||
|
||||
Args:
|
||||
factor (Factor): The factor object associated with the KnowledgeManagement.
|
||||
|
||||
Returns:
|
||||
None
|
||||
"""
|
||||
self.target_task = target_task
|
||||
self.implementation = implementation.copy()
|
||||
self.feedback = feedback
|
||||
|
||||
def get_implementation_and_feedback_str(self) -> str:
|
||||
return f"""------------------Factor implementation code:------------------
|
||||
{self.implementation.code}
|
||||
------------------Factor implementation feedback:------------------
|
||||
return f"""------------------implementation code:------------------
|
||||
{self.implementation.all_codes}
|
||||
------------------implementation feedback:------------------
|
||||
{self.feedback!s}
|
||||
"""
|
||||
|
||||
|
||||
class FactorQueriedKnowledge(QueriedKnowledge):
|
||||
class CoSTEERQueriedKnowledge(QueriedKnowledge):
|
||||
def __init__(self, success_task_to_knowledge_dict: dict = {}, failed_task_info_set: set = set()) -> None:
|
||||
self.success_task_to_knowledge_dict = success_task_to_knowledge_dict
|
||||
self.failed_task_info_set = failed_task_info_set
|
||||
|
||||
|
||||
class FactorKnowledgeBaseV1(EvolvingKnowledgeBase):
|
||||
class CoSTEERKnowledgeBaseV1(EvolvingKnowledgeBase):
|
||||
def __init__(self, path: str | Path = None) -> None:
|
||||
self.implementation_trace: dict[str, FactorKnowledge] = dict()
|
||||
self.implementation_trace: dict[str, CoSTEERKnowledge] = dict()
|
||||
self.success_task_info_set: set[str] = set()
|
||||
|
||||
self.task_to_embedding = dict()
|
||||
super().__init__(path)
|
||||
|
||||
def query(self) -> QueriedKnowledge | None:
|
||||
def query(self) -> CoSTEERQueriedKnowledge | None:
|
||||
"""
|
||||
Query the knowledge base to get the queried knowledge. So far is handled in RAG strategy.
|
||||
"""
|
||||
raise NotImplementedError
|
||||
|
||||
|
||||
class FactorQueriedKnowledgeV1(FactorQueriedKnowledge):
|
||||
def __init__(self) -> None:
|
||||
self.working_task_to_former_failed_knowledge_dict = dict()
|
||||
self.working_task_to_similar_successful_knowledge_dict = dict()
|
||||
super().__init__()
|
||||
class CoSTEERQueriedKnowledgeV1(CoSTEERQueriedKnowledge):
|
||||
def __init__(
|
||||
self,
|
||||
*args,
|
||||
task_to_former_failed_traces: dict = {},
|
||||
task_to_similar_task_successful_knowledge: dict = {},
|
||||
**kwargs,
|
||||
) -> None:
|
||||
self.task_to_former_failed_traces = task_to_former_failed_traces
|
||||
self.task_to_similar_task_successful_knowledge = task_to_similar_task_successful_knowledge
|
||||
super().__init__(*args, **kwargs)
|
||||
|
||||
|
||||
class FactorRAGStrategyV1(RAGStrategy):
|
||||
def __init__(self, knowledgebase: FactorKnowledgeBaseV1) -> None:
|
||||
class CoSTEERRAGStrategyV1(RAGStrategy):
|
||||
def __init__(self, knowledgebase: CoSTEERKnowledgeBaseV1, settings: CoSTEERSettings) -> None:
|
||||
super().__init__(knowledgebase)
|
||||
self.current_generated_trace_count = 0
|
||||
self.settings = settings
|
||||
|
||||
def generate_knowledge(
|
||||
self,
|
||||
@@ -103,6 +97,9 @@ class FactorRAGStrategyV1(RAGStrategy):
|
||||
*,
|
||||
return_knowledge: bool = False,
|
||||
) -> Knowledge | None:
|
||||
raise NotImplementedError(
|
||||
"This method should be considered as an un-implemented method because we encourage everyone to use v2."
|
||||
)
|
||||
if len(evolving_trace) == self.current_generated_trace_count:
|
||||
return
|
||||
else:
|
||||
@@ -120,7 +117,7 @@ class FactorRAGStrategyV1(RAGStrategy):
|
||||
single_feedback = feedback[task_index]
|
||||
if single_feedback is None:
|
||||
continue
|
||||
single_knowledge = FactorKnowledge(
|
||||
single_knowledge = CoSTEERKnowledge(
|
||||
target_task=target_task,
|
||||
implementation=implementation,
|
||||
feedback=single_feedback,
|
||||
@@ -141,43 +138,44 @@ class FactorRAGStrategyV1(RAGStrategy):
|
||||
self,
|
||||
evo: EvolvableSubjects,
|
||||
evolving_trace: list[EvoStep],
|
||||
) -> QueriedKnowledge | None:
|
||||
v1_query_former_trace_limit = FACTOR_IMPLEMENT_SETTINGS.v1_query_former_trace_limit
|
||||
v1_query_similar_success_limit = FACTOR_IMPLEMENT_SETTINGS.v1_query_similar_success_limit
|
||||
fail_task_trial_limit = FACTOR_IMPLEMENT_SETTINGS.fail_task_trial_limit
|
||||
) -> CoSTEERQueriedKnowledge | None:
|
||||
raise NotImplementedError(
|
||||
"This method should be considered as an un-implemented method because we encourage everyone to use v2."
|
||||
)
|
||||
v1_query_former_trace_limit = self.settings.v1_query_former_trace_limit
|
||||
v1_query_similar_success_limit = self.settings.v1_query_similar_success_limit
|
||||
fail_task_trial_limit = self.settings.fail_task_trial_limit
|
||||
|
||||
queried_knowledge = FactorQueriedKnowledgeV1()
|
||||
for target_factor_task in evo.sub_tasks:
|
||||
target_factor_task_information = target_factor_task.get_task_information()
|
||||
if target_factor_task_information in self.knowledgebase.success_task_info_set:
|
||||
queried_knowledge.success_task_to_knowledge_dict[
|
||||
target_factor_task_information
|
||||
] = self.knowledgebase.implementation_trace[target_factor_task_information][-1]
|
||||
queried_knowledge = CoSTEERQueriedKnowledgeV1()
|
||||
for target_task in evo.sub_tasks:
|
||||
target_task_information = target_task.get_task_information()
|
||||
if target_task_information in self.knowledgebase.success_task_info_set:
|
||||
queried_knowledge.success_task_to_knowledge_dict[target_task_information] = (
|
||||
self.knowledgebase.implementation_trace[target_task_information][-1]
|
||||
)
|
||||
elif (
|
||||
len(
|
||||
self.knowledgebase.implementation_trace.setdefault(
|
||||
target_factor_task_information,
|
||||
target_task_information,
|
||||
[],
|
||||
),
|
||||
)
|
||||
>= fail_task_trial_limit
|
||||
):
|
||||
queried_knowledge.failed_task_info_set.add(target_factor_task_information)
|
||||
queried_knowledge.failed_task_info_set.add(target_task_information)
|
||||
else:
|
||||
queried_knowledge.working_task_to_former_failed_knowledge_dict[
|
||||
target_factor_task_information
|
||||
] = self.knowledgebase.implementation_trace.setdefault(
|
||||
target_factor_task_information,
|
||||
[],
|
||||
)[
|
||||
-v1_query_former_trace_limit:
|
||||
]
|
||||
queried_knowledge.task_to_former_failed_traces[target_task_information] = (
|
||||
self.knowledgebase.implementation_trace.setdefault(
|
||||
target_task_information,
|
||||
[],
|
||||
)[-v1_query_former_trace_limit:]
|
||||
)
|
||||
|
||||
knowledge_base_success_task_list = list(
|
||||
self.knowledgebase.success_task_info_set,
|
||||
)
|
||||
similarity = calculate_embedding_distance_between_str_list(
|
||||
[target_factor_task_information],
|
||||
[target_task_information],
|
||||
knowledge_base_success_task_list,
|
||||
)[0]
|
||||
similar_indexes = sorted(
|
||||
@@ -192,33 +190,34 @@ class FactorRAGStrategyV1(RAGStrategy):
|
||||
)[-1]
|
||||
for index in similar_indexes
|
||||
]
|
||||
queried_knowledge.working_task_to_similar_successful_knowledge_dict[
|
||||
target_factor_task_information
|
||||
] = similar_successful_knowledge
|
||||
queried_knowledge.task_to_similar_task_successful_knowledge[target_task_information] = (
|
||||
similar_successful_knowledge
|
||||
)
|
||||
return queried_knowledge
|
||||
|
||||
|
||||
class FactorQueriedGraphKnowledge(FactorQueriedKnowledge):
|
||||
class CoSTEERQueriedKnowledgeV2(CoSTEERQueriedKnowledgeV1):
|
||||
# Aggregation of knowledge
|
||||
def __init__(
|
||||
self,
|
||||
former_traces: dict = {},
|
||||
component_with_success_task: dict = {},
|
||||
error_with_success_task: dict = {},
|
||||
task_to_former_failed_traces: dict = {},
|
||||
task_to_similar_task_successful_knowledge: dict = {},
|
||||
task_to_similar_error_successful_knowledge: dict = {},
|
||||
**kwargs,
|
||||
) -> None:
|
||||
self.former_traces = former_traces
|
||||
self.component_with_success_task = component_with_success_task
|
||||
self.error_with_success_task = error_with_success_task
|
||||
super().__init__(**kwargs)
|
||||
self.task_to_similar_error_successful_knowledge = task_to_similar_error_successful_knowledge
|
||||
super().__init__(
|
||||
task_to_former_failed_traces=task_to_former_failed_traces,
|
||||
task_to_similar_task_successful_knowledge=task_to_similar_task_successful_knowledge,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
|
||||
class FactorGraphRAGStrategy(RAGStrategy):
|
||||
prompt = Prompts(file_path=Path(__file__).parent.parent / "prompts.yaml")
|
||||
|
||||
def __init__(self, knowledgebase: FactorGraphKnowledgeBase) -> None:
|
||||
class CoSTEERRAGStrategyV2(RAGStrategy):
|
||||
def __init__(self, knowledgebase: CoSTEERKnowledgeBaseV2, settings: CoSTEERSettings) -> None:
|
||||
super().__init__(knowledgebase)
|
||||
self.current_generated_trace_count = 0
|
||||
self.settings = settings
|
||||
|
||||
def generate_knowledge(
|
||||
self,
|
||||
@@ -235,14 +234,13 @@ class FactorGraphRAGStrategy(RAGStrategy):
|
||||
implementations = evo_step.evolvable_subjects
|
||||
feedback = evo_step.feedback
|
||||
for task_index in range(len(implementations.sub_tasks)):
|
||||
single_feedback = feedback[task_index]
|
||||
target_task = implementations.sub_tasks[task_index]
|
||||
target_task_information = target_task.get_task_information()
|
||||
implementation = implementations.sub_workspace_list[task_index]
|
||||
single_feedback = feedback[task_index]
|
||||
if single_feedback is None:
|
||||
single_feedback: CoSTEERSingleFeedback = feedback[task_index]
|
||||
if implementation is None or single_feedback is None:
|
||||
continue
|
||||
single_knowledge = FactorKnowledge(
|
||||
single_knowledge = CoSTEERKnowledge(
|
||||
target_task=target_task,
|
||||
implementation=implementation,
|
||||
feedback=single_feedback,
|
||||
@@ -266,15 +264,15 @@ class FactorGraphRAGStrategy(RAGStrategy):
|
||||
else:
|
||||
# generate error node and store into knowledge base
|
||||
error_analysis_result = []
|
||||
if not single_feedback.value_generated_flag:
|
||||
if single_feedback.return_checking:
|
||||
error_analysis_result = self.analyze_error(
|
||||
single_feedback.execution_feedback,
|
||||
feedback_type="execution",
|
||||
single_feedback.return_checking,
|
||||
feedback_type="value",
|
||||
)
|
||||
else:
|
||||
error_analysis_result = self.analyze_error(
|
||||
single_feedback.factor_value_feedback,
|
||||
feedback_type="value",
|
||||
single_feedback.execution,
|
||||
feedback_type="execution",
|
||||
)
|
||||
self.knowledgebase.working_trace_error_analysis.setdefault(
|
||||
target_task_information,
|
||||
@@ -286,35 +284,35 @@ class FactorGraphRAGStrategy(RAGStrategy):
|
||||
self.current_generated_trace_count = len(evolving_trace)
|
||||
return None
|
||||
|
||||
def query(self, evo: EvolvableSubjects, evolving_trace: list[EvoStep]) -> QueriedKnowledge | None:
|
||||
conf_knowledge_sampler = FACTOR_IMPLEMENT_SETTINGS.v2_knowledge_sampler
|
||||
factor_implementation_queried_graph_knowledge = FactorQueriedGraphKnowledge(
|
||||
def query(self, evo: EvolvableSubjects, evolving_trace: list[EvoStep]) -> CoSTEERQueriedKnowledge | None:
|
||||
conf_knowledge_sampler = self.settings.v2_knowledge_sampler
|
||||
queried_knowledge_v2 = CoSTEERQueriedKnowledgeV2(
|
||||
success_task_to_knowledge_dict=self.knowledgebase.success_task_to_knowledge_dict,
|
||||
)
|
||||
|
||||
factor_implementation_queried_graph_knowledge = self.former_trace_query(
|
||||
queried_knowledge_v2 = self.former_trace_query(
|
||||
evo,
|
||||
factor_implementation_queried_graph_knowledge,
|
||||
FACTOR_IMPLEMENT_SETTINGS.v2_query_former_trace_limit,
|
||||
FACTOR_IMPLEMENT_SETTINGS.v2_add_fail_attempt_to_latest_successful_execution,
|
||||
queried_knowledge_v2,
|
||||
self.settings.v2_query_former_trace_limit,
|
||||
self.settings.v2_add_fail_attempt_to_latest_successful_execution,
|
||||
)
|
||||
factor_implementation_queried_graph_knowledge = self.component_query(
|
||||
queried_knowledge_v2 = self.component_query(
|
||||
evo,
|
||||
factor_implementation_queried_graph_knowledge,
|
||||
FACTOR_IMPLEMENT_SETTINGS.v2_query_component_limit,
|
||||
queried_knowledge_v2,
|
||||
self.settings.v2_query_component_limit,
|
||||
knowledge_sampler=conf_knowledge_sampler,
|
||||
)
|
||||
factor_implementation_queried_graph_knowledge = self.error_query(
|
||||
queried_knowledge_v2 = self.error_query(
|
||||
evo,
|
||||
factor_implementation_queried_graph_knowledge,
|
||||
FACTOR_IMPLEMENT_SETTINGS.v2_query_error_limit,
|
||||
queried_knowledge_v2,
|
||||
self.settings.v2_query_error_limit,
|
||||
knowledge_sampler=conf_knowledge_sampler,
|
||||
)
|
||||
return factor_implementation_queried_graph_knowledge
|
||||
return queried_knowledge_v2
|
||||
|
||||
def analyze_component(
|
||||
self,
|
||||
target_factor_task_information,
|
||||
target_task_information,
|
||||
) -> list[UndirectedNode]: # Hardcode: certain component nodes
|
||||
all_component_nodes = self.knowledgebase.graph.get_all_nodes_by_label_list(["component"])
|
||||
if not len(all_component_nodes):
|
||||
@@ -322,21 +320,18 @@ class FactorGraphRAGStrategy(RAGStrategy):
|
||||
all_component_content = ""
|
||||
for _, component_node in enumerate(all_component_nodes):
|
||||
all_component_content += f"{component_node.content}, \n"
|
||||
analyze_component_system_prompt = (
|
||||
Environment(undefined=StrictUndefined)
|
||||
.from_string(self.prompt["analyze_component_prompt_v1_system"])
|
||||
.render(
|
||||
all_component_content=all_component_content,
|
||||
)
|
||||
analyze_component_system_prompt = T(".prompts:analyze_component_prompt_v1_system").r(
|
||||
all_component_content=all_component_content,
|
||||
)
|
||||
|
||||
analyze_component_user_prompt = target_factor_task_information
|
||||
analyze_component_user_prompt = target_task_information
|
||||
try:
|
||||
component_no_list = json.loads(
|
||||
APIBackend().build_messages_and_create_chat_completion(
|
||||
system_prompt=analyze_component_system_prompt,
|
||||
user_prompt=analyze_component_user_prompt,
|
||||
json_mode=True,
|
||||
json_target_type=List[int],
|
||||
),
|
||||
)["component_no_list"]
|
||||
return [all_component_nodes[index - 1] for index in sorted(list(set(component_no_list)))]
|
||||
@@ -391,41 +386,39 @@ class FactorGraphRAGStrategy(RAGStrategy):
|
||||
def former_trace_query(
|
||||
self,
|
||||
evo: EvolvableSubjects,
|
||||
factor_implementation_queried_graph_knowledge: FactorQueriedGraphKnowledge,
|
||||
queried_knowledge_v2: CoSTEERQueriedKnowledgeV2,
|
||||
v2_query_former_trace_limit: int = 5,
|
||||
v2_add_fail_attempt_to_latest_successful_execution: bool = False,
|
||||
) -> Union[QueriedKnowledge, set]:
|
||||
) -> Union[CoSTEERQueriedKnowledge, set]:
|
||||
"""
|
||||
Query the former trace knowledge of the working trace, and find all the failed task information which tried more than fail_task_trial_limit times
|
||||
"""
|
||||
fail_task_trial_limit = FACTOR_IMPLEMENT_SETTINGS.fail_task_trial_limit
|
||||
fail_task_trial_limit = self.settings.fail_task_trial_limit
|
||||
|
||||
for target_factor_task in evo.sub_tasks:
|
||||
target_factor_task_information = target_factor_task.get_task_information()
|
||||
for target_task in evo.sub_tasks:
|
||||
target_task_information = target_task.get_task_information()
|
||||
if (
|
||||
target_factor_task_information not in self.knowledgebase.success_task_to_knowledge_dict
|
||||
and target_factor_task_information in self.knowledgebase.working_trace_knowledge
|
||||
and len(self.knowledgebase.working_trace_knowledge[target_factor_task_information])
|
||||
>= fail_task_trial_limit
|
||||
target_task_information not in self.knowledgebase.success_task_to_knowledge_dict
|
||||
and target_task_information in self.knowledgebase.working_trace_knowledge
|
||||
and len(self.knowledgebase.working_trace_knowledge[target_task_information]) >= fail_task_trial_limit
|
||||
):
|
||||
factor_implementation_queried_graph_knowledge.failed_task_info_set.add(target_factor_task_information)
|
||||
queried_knowledge_v2.failed_task_info_set.add(target_task_information)
|
||||
|
||||
if (
|
||||
target_factor_task_information not in self.knowledgebase.success_task_to_knowledge_dict
|
||||
and target_factor_task_information
|
||||
not in factor_implementation_queried_graph_knowledge.failed_task_info_set
|
||||
and target_factor_task_information in self.knowledgebase.working_trace_knowledge
|
||||
target_task_information not in self.knowledgebase.success_task_to_knowledge_dict
|
||||
and target_task_information not in queried_knowledge_v2.failed_task_info_set
|
||||
and target_task_information in self.knowledgebase.working_trace_knowledge
|
||||
):
|
||||
former_trace_knowledge = copy.copy(
|
||||
self.knowledgebase.working_trace_knowledge[target_factor_task_information],
|
||||
self.knowledgebase.working_trace_knowledge[target_task_information],
|
||||
)
|
||||
# in former trace query we will delete the right trace in the following order:[..., value_generated_flag is True, value_generated_flag is False, ...]
|
||||
# because we think this order means a deterioration of the trial (like a wrong gradient descent)
|
||||
current_index = 1
|
||||
while current_index < len(former_trace_knowledge):
|
||||
if (
|
||||
not former_trace_knowledge[current_index].feedback.value_generated_flag
|
||||
and former_trace_knowledge[current_index - 1].feedback.value_generated_flag
|
||||
not former_trace_knowledge[current_index].feedback.return_checking
|
||||
and former_trace_knowledge[current_index - 1].feedback.return_checking
|
||||
):
|
||||
former_trace_knowledge.pop(current_index)
|
||||
else:
|
||||
@@ -436,47 +429,44 @@ class FactorGraphRAGStrategy(RAGStrategy):
|
||||
# When the last successful execution is not the last one in the working trace, it means we have tried to correct it. We should tell the agent this fail trial to avoid endless loop in the future.
|
||||
if (
|
||||
len(former_trace_knowledge) > 0
|
||||
and len(self.knowledgebase.working_trace_knowledge[target_factor_task_information]) > 1
|
||||
and self.knowledgebase.working_trace_knowledge[target_factor_task_information].index(
|
||||
and len(self.knowledgebase.working_trace_knowledge[target_task_information]) > 1
|
||||
and self.knowledgebase.working_trace_knowledge[target_task_information].index(
|
||||
former_trace_knowledge[-1]
|
||||
)
|
||||
< len(self.knowledgebase.working_trace_knowledge[target_factor_task_information]) - 1
|
||||
< len(self.knowledgebase.working_trace_knowledge[target_task_information]) - 1
|
||||
):
|
||||
latest_attempt = self.knowledgebase.working_trace_knowledge[target_factor_task_information][-1]
|
||||
latest_attempt = self.knowledgebase.working_trace_knowledge[target_task_information][-1]
|
||||
|
||||
factor_implementation_queried_graph_knowledge.former_traces[target_factor_task_information] = (
|
||||
queried_knowledge_v2.task_to_former_failed_traces[target_task_information] = (
|
||||
former_trace_knowledge[-v2_query_former_trace_limit:],
|
||||
latest_attempt,
|
||||
)
|
||||
else:
|
||||
factor_implementation_queried_graph_knowledge.former_traces[target_factor_task_information] = ([], None)
|
||||
queried_knowledge_v2.task_to_former_failed_traces[target_task_information] = ([], None)
|
||||
|
||||
return factor_implementation_queried_graph_knowledge
|
||||
return queried_knowledge_v2
|
||||
|
||||
def component_query(
|
||||
self,
|
||||
evo: EvolvableSubjects,
|
||||
factor_implementation_queried_graph_knowledge: FactorQueriedGraphKnowledge,
|
||||
queried_knowledge_v2: CoSTEERQueriedKnowledgeV2,
|
||||
v2_query_component_limit: int = 5,
|
||||
knowledge_sampler: float = 1.0,
|
||||
) -> QueriedKnowledge | None:
|
||||
# queried_component_knowledge = FactorQueriedGraphComponentKnowledge()
|
||||
for target_factor_task in evo.sub_tasks:
|
||||
target_factor_task_information = target_factor_task.get_task_information()
|
||||
) -> CoSTEERQueriedKnowledge | None:
|
||||
for target_task in evo.sub_tasks:
|
||||
target_task_information = target_task.get_task_information()
|
||||
if (
|
||||
target_factor_task_information in self.knowledgebase.success_task_to_knowledge_dict
|
||||
or target_factor_task_information in factor_implementation_queried_graph_knowledge.failed_task_info_set
|
||||
target_task_information in self.knowledgebase.success_task_to_knowledge_dict
|
||||
or target_task_information in queried_knowledge_v2.failed_task_info_set
|
||||
):
|
||||
factor_implementation_queried_graph_knowledge.component_with_success_task[
|
||||
target_factor_task_information
|
||||
] = []
|
||||
queried_knowledge_v2.task_to_similar_task_successful_knowledge[target_task_information] = []
|
||||
else:
|
||||
if target_factor_task_information not in self.knowledgebase.task_to_component_nodes:
|
||||
self.knowledgebase.task_to_component_nodes[target_factor_task_information] = self.analyze_component(
|
||||
target_factor_task_information,
|
||||
if target_task_information not in self.knowledgebase.task_to_component_nodes:
|
||||
self.knowledgebase.task_to_component_nodes[target_task_information] = self.analyze_component(
|
||||
target_task_information,
|
||||
)
|
||||
|
||||
component_analysis_result = self.knowledgebase.task_to_component_nodes[target_factor_task_information]
|
||||
component_analysis_result = self.knowledgebase.task_to_component_nodes[target_task_information]
|
||||
|
||||
if len(component_analysis_result) > 1:
|
||||
task_des_node_list = self.knowledgebase.graph_query_by_intersection(
|
||||
@@ -487,9 +477,7 @@ class FactorGraphRAGStrategy(RAGStrategy):
|
||||
else:
|
||||
task_des_node_list = []
|
||||
single_component_constraint = v2_query_component_limit
|
||||
factor_implementation_queried_graph_knowledge.component_with_success_task[
|
||||
target_factor_task_information
|
||||
] = []
|
||||
queried_knowledge_v2.task_to_similar_task_successful_knowledge[target_task_information] = []
|
||||
for component_node in component_analysis_result:
|
||||
# Reverse iterate, a trade-off with intersection search
|
||||
count = 0
|
||||
@@ -520,19 +508,19 @@ class FactorGraphRAGStrategy(RAGStrategy):
|
||||
]
|
||||
if (
|
||||
target_knowledge
|
||||
not in factor_implementation_queried_graph_knowledge.component_with_success_task[
|
||||
target_factor_task_information
|
||||
not in queried_knowledge_v2.task_to_similar_task_successful_knowledge[
|
||||
target_task_information
|
||||
]
|
||||
):
|
||||
factor_implementation_queried_graph_knowledge.component_with_success_task[
|
||||
target_factor_task_information
|
||||
queried_knowledge_v2.task_to_similar_task_successful_knowledge[
|
||||
target_task_information
|
||||
].append(target_knowledge)
|
||||
|
||||
# finally add embedding related knowledge
|
||||
knowledge_base_success_task_list = list(self.knowledgebase.success_task_to_knowledge_dict)
|
||||
|
||||
similarity = calculate_embedding_distance_between_str_list(
|
||||
[target_factor_task_information],
|
||||
[target_task_information],
|
||||
knowledge_base_success_task_list,
|
||||
)[0]
|
||||
similar_indexes = sorted(
|
||||
@@ -547,28 +535,24 @@ class FactorGraphRAGStrategy(RAGStrategy):
|
||||
for knowledge in embedding_similar_successful_knowledge:
|
||||
if (
|
||||
knowledge
|
||||
not in factor_implementation_queried_graph_knowledge.component_with_success_task[
|
||||
target_factor_task_information
|
||||
]
|
||||
not in queried_knowledge_v2.task_to_similar_task_successful_knowledge[target_task_information]
|
||||
):
|
||||
factor_implementation_queried_graph_knowledge.component_with_success_task[
|
||||
target_factor_task_information
|
||||
].append(knowledge)
|
||||
queried_knowledge_v2.task_to_similar_task_successful_knowledge[target_task_information].append(
|
||||
knowledge
|
||||
)
|
||||
|
||||
if knowledge_sampler > 0:
|
||||
factor_implementation_queried_graph_knowledge.component_with_success_task[
|
||||
target_factor_task_information
|
||||
] = [
|
||||
queried_knowledge_v2.task_to_similar_task_successful_knowledge[target_task_information] = [
|
||||
knowledge
|
||||
for knowledge in factor_implementation_queried_graph_knowledge.component_with_success_task[
|
||||
target_factor_task_information
|
||||
for knowledge in queried_knowledge_v2.task_to_similar_task_successful_knowledge[
|
||||
target_task_information
|
||||
]
|
||||
if random.uniform(0, 1) <= knowledge_sampler
|
||||
]
|
||||
|
||||
# Make sure no less than half of the knowledge are from GT
|
||||
queried_knowledge_list = factor_implementation_queried_graph_knowledge.component_with_success_task[
|
||||
target_factor_task_information
|
||||
queried_knowledge_list = queried_knowledge_v2.task_to_similar_task_successful_knowledge[
|
||||
target_task_information
|
||||
]
|
||||
queried_from_gt_knowledge_list = [
|
||||
knowledge
|
||||
@@ -584,51 +568,43 @@ class FactorGraphRAGStrategy(RAGStrategy):
|
||||
min((v2_query_component_limit // 2 + 1), len(queried_from_gt_knowledge_list)),
|
||||
v2_query_component_limit - len(queried_without_gt_knowledge_list),
|
||||
)
|
||||
factor_implementation_queried_graph_knowledge.component_with_success_task[
|
||||
target_factor_task_information
|
||||
] = (
|
||||
queried_knowledge_v2.task_to_similar_task_successful_knowledge[target_task_information] = (
|
||||
queried_from_gt_knowledge_list[:queried_from_gt_knowledge_count]
|
||||
+ queried_without_gt_knowledge_list[: v2_query_component_limit - queried_from_gt_knowledge_count]
|
||||
)
|
||||
|
||||
return factor_implementation_queried_graph_knowledge
|
||||
return queried_knowledge_v2
|
||||
|
||||
def error_query(
|
||||
self,
|
||||
evo: EvolvableSubjects,
|
||||
factor_implementation_queried_graph_knowledge: FactorQueriedGraphKnowledge,
|
||||
queried_knowledge_v2: CoSTEERQueriedKnowledgeV2,
|
||||
v2_query_error_limit: int = 5,
|
||||
knowledge_sampler: float = 1.0,
|
||||
) -> QueriedKnowledge | None:
|
||||
# queried_error_knowledge = FactorQueriedGraphErrorKnowledge()
|
||||
for task_index, target_factor_task in enumerate(evo.sub_tasks):
|
||||
target_factor_task_information = target_factor_task.get_task_information()
|
||||
factor_implementation_queried_graph_knowledge.error_with_success_task[target_factor_task_information] = {}
|
||||
) -> CoSTEERQueriedKnowledge | None:
|
||||
for task_index, target_task in enumerate(evo.sub_tasks):
|
||||
target_task_information = target_task.get_task_information()
|
||||
queried_knowledge_v2.task_to_similar_error_successful_knowledge[target_task_information] = []
|
||||
if (
|
||||
target_factor_task_information in self.knowledgebase.success_task_to_knowledge_dict
|
||||
or target_factor_task_information in factor_implementation_queried_graph_knowledge.failed_task_info_set
|
||||
target_task_information in self.knowledgebase.success_task_to_knowledge_dict
|
||||
or target_task_information in queried_knowledge_v2.failed_task_info_set
|
||||
):
|
||||
factor_implementation_queried_graph_knowledge.error_with_success_task[
|
||||
target_factor_task_information
|
||||
] = []
|
||||
queried_knowledge_v2.task_to_similar_error_successful_knowledge[target_task_information] = []
|
||||
else:
|
||||
factor_implementation_queried_graph_knowledge.error_with_success_task[
|
||||
target_factor_task_information
|
||||
] = []
|
||||
queried_knowledge_v2.task_to_similar_error_successful_knowledge[target_task_information] = []
|
||||
if (
|
||||
target_factor_task_information in self.knowledgebase.working_trace_error_analysis
|
||||
and len(self.knowledgebase.working_trace_error_analysis[target_factor_task_information]) > 0
|
||||
and len(factor_implementation_queried_graph_knowledge.former_traces[target_factor_task_information])
|
||||
> 0
|
||||
target_task_information in self.knowledgebase.working_trace_error_analysis
|
||||
and len(self.knowledgebase.working_trace_error_analysis[target_task_information]) > 0
|
||||
and len(queried_knowledge_v2.task_to_former_failed_traces[target_task_information]) > 0
|
||||
):
|
||||
queried_last_trace = factor_implementation_queried_graph_knowledge.former_traces[
|
||||
target_factor_task_information
|
||||
][0][-1]
|
||||
target_index = self.knowledgebase.working_trace_knowledge[target_factor_task_information].index(
|
||||
queried_last_trace = queried_knowledge_v2.task_to_former_failed_traces[target_task_information][0][
|
||||
-1
|
||||
]
|
||||
target_index = self.knowledgebase.working_trace_knowledge[target_task_information].index(
|
||||
queried_last_trace,
|
||||
)
|
||||
last_knowledge_error_analysis_result = self.knowledgebase.working_trace_error_analysis[
|
||||
target_factor_task_information
|
||||
target_task_information
|
||||
][target_index]
|
||||
else:
|
||||
last_knowledge_error_analysis_result = []
|
||||
@@ -721,20 +697,20 @@ class FactorGraphRAGStrategy(RAGStrategy):
|
||||
]
|
||||
|
||||
same_error_success_knowledge_pair_list = same_error_success_knowledge_pair_list[:v2_query_error_limit]
|
||||
factor_implementation_queried_graph_knowledge.error_with_success_task[
|
||||
target_factor_task_information
|
||||
] = same_error_success_knowledge_pair_list
|
||||
queried_knowledge_v2.task_to_similar_error_successful_knowledge[target_task_information] = (
|
||||
same_error_success_knowledge_pair_list
|
||||
)
|
||||
|
||||
return factor_implementation_queried_graph_knowledge
|
||||
return queried_knowledge_v2
|
||||
|
||||
|
||||
class FactorGraphKnowledgeBase(EvolvingKnowledgeBase):
|
||||
class CoSTEERKnowledgeBaseV2(EvolvingKnowledgeBase):
|
||||
def __init__(self, init_component_list=None, path: str | Path = None) -> None:
|
||||
"""
|
||||
Load knowledge, offer brief information of knowledge and common handle interfaces
|
||||
"""
|
||||
self.graph: UndirectedGraph = UndirectedGraph(Path.cwd() / "graph.pkl")
|
||||
logger.info(f"Knowledge Graph loaded, size={self.graph.size()}")
|
||||
logger.info(f"CoSTEER Knowledge Graph loaded, size={self.graph.size()}")
|
||||
|
||||
if init_component_list:
|
||||
for component in init_component_list:
|
||||
@@ -751,7 +727,7 @@ class FactorGraphKnowledgeBase(EvolvingKnowledgeBase):
|
||||
# Add already success task
|
||||
self.success_task_to_knowledge_dict = {}
|
||||
|
||||
# key:node_id(for task trace and success implement), value:knowledge instance(aka 'FactorKnowledge')
|
||||
# key:node_id(for task trace and success implement), value:knowledge instance(aka 'CoSTEERKnowledge')
|
||||
self.node_to_implementation_knowledge_dict = {}
|
||||
|
||||
# store the task description to component nodes
|
||||
@@ -0,0 +1,10 @@
|
||||
|
||||
analyze_component_prompt_v1_system: |-
|
||||
User is getting a new task that might consist of the components below (given in component_index: component_description):
|
||||
{{all_component_content}}
|
||||
|
||||
You should find out what components does the new task have, and put their indices in a list.
|
||||
Please response the critic in the json format. Here is an example structure for the JSON output, please strictly follow the format:
|
||||
{
|
||||
"component_no_list": the list containing indices of components.
|
||||
}
|
||||
@@ -0,0 +1,9 @@
|
||||
from rdagent.core.experiment import Task
|
||||
|
||||
|
||||
class CoSTEERTask(Task):
|
||||
def __init__(self, base_code: str = None, *args, **kwargs) -> None:
|
||||
super().__init__(*args, **kwargs)
|
||||
# TODO: we may upgrade the base_code into a workspace-like thing to know previous.
|
||||
# NOTE: (xiao) think we don't need the base_code anymore. The information should be retrieved from the workspace.
|
||||
self.base_code = base_code
|
||||
@@ -0,0 +1,71 @@
|
||||
from typing import Literal
|
||||
|
||||
from rdagent.app.data_science.conf import DS_RD_SETTING
|
||||
from rdagent.components.coder.CoSTEER.config import CoSTEERSettings
|
||||
from rdagent.utils.env import (
|
||||
CondaConf,
|
||||
DockerEnv,
|
||||
DSDockerConf,
|
||||
Env,
|
||||
LocalEnv,
|
||||
MLEBDockerConf,
|
||||
MLECondaConf,
|
||||
)
|
||||
|
||||
|
||||
class DSCoderCoSTEERSettings(CoSTEERSettings):
|
||||
"""Data Science CoSTEER settings"""
|
||||
|
||||
class Config:
|
||||
env_prefix = "DS_Coder_CoSTEER_"
|
||||
|
||||
max_seconds: int = 2400
|
||||
env_type: str = "docker"
|
||||
# TODO: extract a function for env and conf.
|
||||
|
||||
|
||||
def get_ds_env(
|
||||
conf_type: Literal["kaggle", "mlebench"] = "kaggle",
|
||||
extra_volumes: dict = {},
|
||||
running_timeout_period: int = (
|
||||
DS_RD_SETTING.debug_timeout if DS_RD_SETTING.sample_data else DS_RD_SETTING.full_timeout
|
||||
),
|
||||
) -> Env:
|
||||
"""
|
||||
Retrieve the appropriate environment configuration based on the env_type setting.
|
||||
|
||||
Returns:
|
||||
Env: An instance of the environment configured either as DockerEnv or LocalEnv.
|
||||
|
||||
Raises:
|
||||
ValueError: If the env_type is not recognized.
|
||||
"""
|
||||
conf = DSCoderCoSTEERSettings()
|
||||
assert conf_type in ["kaggle", "mlebench"], f"Unknown conf_type: {conf_type}"
|
||||
|
||||
if conf.env_type == "docker":
|
||||
env_conf = DSDockerConf() if conf_type == "kaggle" else MLEBDockerConf()
|
||||
env = DockerEnv(conf=env_conf)
|
||||
elif conf.env_type == "conda":
|
||||
env = LocalEnv(
|
||||
conf=(
|
||||
CondaConf(conda_env_name=conf_type) if conf_type == "kaggle" else MLECondaConf(conda_env_name=conf_type)
|
||||
)
|
||||
)
|
||||
else:
|
||||
raise ValueError(f"Unknown env type: {conf.env_type}")
|
||||
env.conf.extra_volumes = extra_volumes
|
||||
env.conf.running_timeout_period = running_timeout_period
|
||||
return env
|
||||
|
||||
|
||||
def get_clear_ws_cmd(stage: Literal["before_training", "before_inference"] = "before_training") -> str:
|
||||
"""
|
||||
Clean the files in workspace to a specific stage
|
||||
"""
|
||||
assert stage in ["before_training", "before_inference"], f"Unknown stage: {stage}"
|
||||
if DS_RD_SETTING.enable_model_dump and stage == "before_training":
|
||||
cmd = "rm -r submission.csv scores.csv models"
|
||||
else:
|
||||
cmd = "rm submission.csv scores.csv"
|
||||
return cmd
|
||||
@@ -0,0 +1,164 @@
|
||||
"""
|
||||
File structure
|
||||
- ___init__.py: the entrance/agent of coder
|
||||
- evaluator.py
|
||||
- conf.py
|
||||
- exp.py: everything under the experiment, e.g.
|
||||
- Task
|
||||
- Experiment
|
||||
- Workspace
|
||||
- test.py
|
||||
- Each coder could be tested.
|
||||
"""
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from jinja2 import Environment, StrictUndefined
|
||||
|
||||
from rdagent.app.data_science.conf import DS_RD_SETTING
|
||||
from rdagent.components.coder.CoSTEER import CoSTEER
|
||||
from rdagent.components.coder.CoSTEER.evaluators import (
|
||||
CoSTEERMultiEvaluator,
|
||||
CoSTEERSingleFeedback,
|
||||
)
|
||||
from rdagent.components.coder.CoSTEER.evolving_strategy import (
|
||||
MultiProcessEvolvingStrategy,
|
||||
)
|
||||
from rdagent.components.coder.CoSTEER.knowledge_management import (
|
||||
CoSTEERQueriedKnowledge,
|
||||
)
|
||||
from rdagent.components.coder.data_science.conf import DSCoderCoSTEERSettings
|
||||
from rdagent.components.coder.data_science.ensemble.eval import EnsembleCoSTEEREvaluator
|
||||
from rdagent.components.coder.data_science.ensemble.exp import EnsembleTask
|
||||
from rdagent.core.exception import CoderError
|
||||
from rdagent.core.experiment import FBWorkspace
|
||||
from rdagent.core.scenario import Scenario
|
||||
from rdagent.oai.llm_utils import APIBackend
|
||||
from rdagent.utils.agent.ret import PythonAgentOut
|
||||
from rdagent.utils.agent.tpl import T
|
||||
|
||||
DIRNAME = Path(__file__).absolute().resolve().parent
|
||||
|
||||
|
||||
class EnsembleMultiProcessEvolvingStrategy(MultiProcessEvolvingStrategy):
|
||||
def implement_one_task(
|
||||
self,
|
||||
target_task: EnsembleTask,
|
||||
queried_knowledge: CoSTEERQueriedKnowledge | None = None,
|
||||
workspace: FBWorkspace | None = None,
|
||||
prev_task_feedback: CoSTEERSingleFeedback | None = None,
|
||||
) -> dict[str, str]:
|
||||
# Get task information for knowledge querying
|
||||
ensemble_information_str = target_task.get_task_information()
|
||||
|
||||
# Query knowledge
|
||||
queried_similar_successful_knowledge = (
|
||||
queried_knowledge.task_to_similar_task_successful_knowledge[ensemble_information_str]
|
||||
if queried_knowledge is not None
|
||||
else []
|
||||
)
|
||||
queried_former_failed_knowledge = (
|
||||
queried_knowledge.task_to_former_failed_traces[ensemble_information_str]
|
||||
if queried_knowledge is not None
|
||||
else []
|
||||
)
|
||||
queried_former_failed_knowledge = (
|
||||
[
|
||||
knowledge
|
||||
for knowledge in queried_former_failed_knowledge[0]
|
||||
if knowledge.implementation.file_dict.get("ensemble.py") != workspace.file_dict.get("ensemble.py")
|
||||
],
|
||||
queried_former_failed_knowledge[1],
|
||||
)
|
||||
|
||||
# Generate code with knowledge integration
|
||||
competition_info = self.scen.get_scenario_all_desc(eda_output=workspace.file_dict.get("EDA.md", None))
|
||||
system_prompt = T(".prompts:ensemble_coder.system").r(
|
||||
task_desc=ensemble_information_str,
|
||||
competition_info=competition_info,
|
||||
queried_similar_successful_knowledge=queried_similar_successful_knowledge,
|
||||
queried_former_failed_knowledge=(
|
||||
queried_former_failed_knowledge[0] if queried_former_failed_knowledge else None
|
||||
),
|
||||
all_code=workspace.all_codes,
|
||||
out_spec=PythonAgentOut.get_spec(),
|
||||
)
|
||||
|
||||
if DS_RD_SETTING.spec_enabled:
|
||||
code_spec = workspace.file_dict["spec/ensemble.md"]
|
||||
else:
|
||||
test_code = (
|
||||
Environment(undefined=StrictUndefined)
|
||||
.from_string((DIRNAME / "eval_tests" / "ensemble_test.txt").read_text())
|
||||
.render(
|
||||
model_names=[
|
||||
fn[:-3] for fn in workspace.file_dict.keys() if fn.startswith("model_") and "test" not in fn
|
||||
],
|
||||
metric_name=self.scen.metric_name,
|
||||
)
|
||||
)
|
||||
code_spec = T("scenarios.data_science.share:component_spec.general").r(
|
||||
spec=T("scenarios.data_science.share:component_spec.Ensemble").r(), test_code=test_code
|
||||
)
|
||||
user_prompt = T(".prompts:ensemble_coder.user").r(
|
||||
code_spec=code_spec,
|
||||
latest_code=workspace.file_dict.get("ensemble.py"),
|
||||
latest_code_feedback=prev_task_feedback,
|
||||
)
|
||||
|
||||
for _ in range(5):
|
||||
ensemble_code = PythonAgentOut.extract_output(
|
||||
APIBackend().build_messages_and_create_chat_completion(
|
||||
user_prompt=user_prompt,
|
||||
system_prompt=system_prompt,
|
||||
)
|
||||
)
|
||||
if ensemble_code != workspace.file_dict.get("ensemble.py"):
|
||||
break
|
||||
else:
|
||||
user_prompt = user_prompt + "\nPlease avoid generating same code to former code!"
|
||||
else:
|
||||
raise CoderError("Failed to generate a new ensemble code.")
|
||||
|
||||
return {
|
||||
"ensemble.py": ensemble_code,
|
||||
}
|
||||
|
||||
def assign_code_list_to_evo(self, code_list: list[dict[str, str]], evo):
|
||||
"""
|
||||
Assign the code list to the evolving item.
|
||||
|
||||
The code list is aligned with the evolving item's sub-tasks.
|
||||
If a task is not implemented, put a None in the list.
|
||||
"""
|
||||
for index in range(len(evo.sub_tasks)):
|
||||
if code_list[index] is None:
|
||||
continue
|
||||
if evo.sub_workspace_list[index] is None:
|
||||
# evo.sub_workspace_list[index] = FBWorkspace(target_task=evo.sub_tasks[index])
|
||||
evo.sub_workspace_list[index] = evo.experiment_workspace
|
||||
evo.sub_workspace_list[index].inject_files(**code_list[index])
|
||||
return evo
|
||||
|
||||
|
||||
class EnsembleCoSTEER(CoSTEER):
|
||||
def __init__(
|
||||
self,
|
||||
scen: Scenario,
|
||||
*args,
|
||||
**kwargs,
|
||||
) -> None:
|
||||
settings = DSCoderCoSTEERSettings()
|
||||
eva = CoSTEERMultiEvaluator(EnsembleCoSTEEREvaluator(scen=scen), scen=scen)
|
||||
es = EnsembleMultiProcessEvolvingStrategy(scen=scen, settings=settings)
|
||||
|
||||
super().__init__(
|
||||
*args,
|
||||
settings=settings,
|
||||
eva=eva,
|
||||
es=es,
|
||||
evolving_version=2,
|
||||
scen=scen,
|
||||
max_loop=DS_RD_SETTING.coder_max_loop,
|
||||
**kwargs,
|
||||
)
|
||||
@@ -0,0 +1,2 @@
|
||||
# Configuration file for ensemble component
|
||||
# Currently empty as no specific configuration is needed
|
||||
@@ -0,0 +1,95 @@
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
from jinja2 import Environment, StrictUndefined
|
||||
|
||||
from rdagent.app.data_science.conf import DS_RD_SETTING
|
||||
from rdagent.components.coder.CoSTEER.evaluators import (
|
||||
CoSTEEREvaluator,
|
||||
CoSTEERSingleFeedback,
|
||||
)
|
||||
from rdagent.components.coder.data_science.conf import get_ds_env
|
||||
from rdagent.components.coder.data_science.utils import remove_eda_part
|
||||
from rdagent.core.evolving_framework import QueriedKnowledge
|
||||
from rdagent.core.experiment import FBWorkspace, Task
|
||||
from rdagent.utils.agent.tpl import T
|
||||
from rdagent.utils.agent.workflow import build_cls_from_json_with_retry
|
||||
|
||||
DIRNAME = Path(__file__).absolute().resolve().parent
|
||||
|
||||
EnsembleEvalFeedback = CoSTEERSingleFeedback
|
||||
|
||||
|
||||
class EnsembleCoSTEEREvaluator(CoSTEEREvaluator):
|
||||
def evaluate(
|
||||
self,
|
||||
target_task: Task,
|
||||
implementation: FBWorkspace,
|
||||
gt_implementation: FBWorkspace,
|
||||
queried_knowledge: QueriedKnowledge = None,
|
||||
**kwargs,
|
||||
) -> EnsembleEvalFeedback:
|
||||
|
||||
target_task_information = target_task.get_task_information()
|
||||
metric_name = self.scen.metric_name
|
||||
|
||||
if (
|
||||
queried_knowledge is not None
|
||||
and target_task_information in queried_knowledge.success_task_to_knowledge_dict
|
||||
):
|
||||
return queried_knowledge.success_task_to_knowledge_dict[target_task_information].feedback
|
||||
elif queried_knowledge is not None and target_task_information in queried_knowledge.failed_task_info_set:
|
||||
return EnsembleEvalFeedback(
|
||||
execution="This task has failed too many times, skip implementation.",
|
||||
code="This task has failed too many times, skip implementation.",
|
||||
return_checking="This task has failed too many times, skip implementation.",
|
||||
final_decision=False,
|
||||
)
|
||||
|
||||
env = get_ds_env(extra_volumes={self.scen.debug_path: T("scenarios.data_science.share:scen.input_path").r()})
|
||||
|
||||
fname = "test/ensemble_test.txt"
|
||||
test_code = (DIRNAME / "eval_tests" / "ensemble_test.txt").read_text()
|
||||
test_code = (
|
||||
Environment(undefined=StrictUndefined)
|
||||
.from_string(test_code)
|
||||
.render(
|
||||
model_names=[
|
||||
fn[:-3] for fn in implementation.file_dict.keys() if fn.startswith("model_") and "test" not in fn
|
||||
],
|
||||
metric_name=metric_name,
|
||||
)
|
||||
)
|
||||
|
||||
implementation.inject_files(**{fname: test_code})
|
||||
stdout, ret_code = implementation.execute_ret_code(env=env, entry=f"python {fname}")
|
||||
|
||||
stdout += f"\nNOTE: the above scripts run with return code {ret_code}"
|
||||
|
||||
if "main.py" in implementation.file_dict and ret_code == 0:
|
||||
workflow_stdout = implementation.execute(env=env, entry="python main.py")
|
||||
workflow_stdout = remove_eda_part(workflow_stdout)
|
||||
else:
|
||||
workflow_stdout = None
|
||||
|
||||
system_prompt = T(".prompts:ensemble_eval.system").r(
|
||||
task_desc=target_task_information,
|
||||
test_code=test_code,
|
||||
metric_name=metric_name,
|
||||
code=implementation.file_dict["ensemble.py"],
|
||||
workflow_stdout=workflow_stdout,
|
||||
workflow_code=implementation.all_codes,
|
||||
)
|
||||
user_prompt = T(".prompts:ensemble_eval.user").r(
|
||||
stdout=stdout,
|
||||
workflow_stdout=workflow_stdout,
|
||||
)
|
||||
efb = build_cls_from_json_with_retry(
|
||||
EnsembleEvalFeedback,
|
||||
system_prompt=system_prompt,
|
||||
user_prompt=user_prompt,
|
||||
init_kwargs_update_func=EnsembleEvalFeedback.val_and_update_init_dict,
|
||||
)
|
||||
efb.final_decision = efb.final_decision and ret_code == 0
|
||||
return efb
|
||||
@@ -0,0 +1,137 @@
|
||||
"""
|
||||
Tests for `ensemble_workflow` in ensemble.py
|
||||
|
||||
A qualified ensemble_workflow implementation should:
|
||||
- Return predictions
|
||||
- Have correct shapes for inputs and outputs
|
||||
- Use validation data appropriately
|
||||
- Generate a scores.csv file
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from pathlib import Path
|
||||
from sklearn.model_selection import train_test_split
|
||||
import torch
|
||||
import tensorflow as tf
|
||||
from load_data import load_data
|
||||
from feature import feat_eng
|
||||
from ensemble import ensemble_workflow
|
||||
|
||||
def print_preds_info(model_name, data_type, preds):
|
||||
if preds is None:
|
||||
print(f"Model {model_name} {data_type} predictions: None")
|
||||
else:
|
||||
print(f"Model {model_name} {data_type} predictions shape: {preds.shape}")
|
||||
|
||||
print("Showing a preview of the predictions (first few entries only):")
|
||||
if isinstance(preds, (pd.DataFrame, pd.Series)):
|
||||
print(preds.head())
|
||||
elif isinstance(preds, (np.ndarray, torch.Tensor, tf.Tensor)):
|
||||
print(preds[:2])
|
||||
elif isinstance(preds, list):
|
||||
print(pd.DataFrame(preds[:5]))
|
||||
else:
|
||||
print(f"Unknown prediction type: {type(preds)}")
|
||||
|
||||
def get_length(data):
|
||||
return data.shape[0] if hasattr(data, 'shape') else len(data)
|
||||
|
||||
X, y, test_X, test_ids = load_data()
|
||||
X, y, test_X = feat_eng(X, y, test_X)
|
||||
train_X, val_X, train_y, val_y = train_test_split(X, y, test_size=0.2, random_state=42)
|
||||
|
||||
# Print the types of train_y and val_y
|
||||
print(f"train_y type: {type(train_y)}, val_y type: {type(val_y)}")
|
||||
|
||||
test_preds_dict = {}
|
||||
val_preds_dict = {}
|
||||
{% for mn in model_names %}
|
||||
from {{mn}} import model_workflow as {{mn}}_workflow
|
||||
val_preds_dict["{{mn}}"], test_preds_dict["{{mn}}"], _ = {{mn}}_workflow(
|
||||
X=train_X,
|
||||
y=train_y,
|
||||
val_X=val_X,
|
||||
val_y=val_y,
|
||||
test_X=test_X
|
||||
)
|
||||
|
||||
print_preds_info("{{mn}}", "test", test_preds_dict["{{mn}}"])
|
||||
{% endfor %}
|
||||
|
||||
for key in val_preds_dict.keys():
|
||||
if val_preds_dict[key] is None:
|
||||
print(f"Model {key} validation predictions (val_preds_dict[key]) is None.")
|
||||
elif isinstance(val_preds_dict[key], list):
|
||||
print(f"Model {key} validation predictions (val_preds_dict[key]) (list type) length: {len(val_preds_dict[key])}")
|
||||
else:
|
||||
print(f"Model {key} validation predictions (val_preds_dict[key]) shape: {val_preds_dict[key].shape}")
|
||||
|
||||
if test_preds_dict[key] is None:
|
||||
print(f"Model {key} test predictions (test_preds_dict[key]) is None.")
|
||||
elif isinstance(test_preds_dict[key], list):
|
||||
print(f"Model {key} test predictions (test_preds_dict[key]) (list type) length: {len(test_preds_dict[key])}")
|
||||
else:
|
||||
print(f"Model {key} test predictions (test_preds_dict[key]) shape: {test_preds_dict[key].shape}")
|
||||
|
||||
print(f"val_y.shape: {val_y.shape}" if not isinstance(val_y, list) else f"val_y(list)'s length: {len(val_y)}")
|
||||
|
||||
import sys
|
||||
import reprlib
|
||||
def debug_info_print(func):
|
||||
aRepr = reprlib.Repr()
|
||||
aRepr.maxother=300
|
||||
def wrapper(*args, **kwargs):
|
||||
def local_trace(frame, event, arg):
|
||||
if event == "return" and frame.f_code == func.__code__:
|
||||
print("\n" + "="*20 + "Running ensemble code, local variable values:" + "="*20)
|
||||
for k, v in frame.f_locals.items():
|
||||
printed = aRepr.repr(v)
|
||||
print(f"{k}:\n {printed}")
|
||||
print("="*20 + "Local variable values end" + "="*20)
|
||||
return local_trace
|
||||
|
||||
sys.settrace(local_trace)
|
||||
try:
|
||||
return func(*args, **kwargs)
|
||||
finally:
|
||||
sys.settrace(None)
|
||||
return wrapper
|
||||
|
||||
|
||||
# Run ensemble
|
||||
final_pred = debug_info_print(ensemble_workflow)(test_preds_dict, val_preds_dict, val_y)
|
||||
|
||||
print_preds_info("ensemble", "test", final_pred)
|
||||
|
||||
# Check type
|
||||
pred_type = type(next(iter(test_preds_dict.values())))
|
||||
assert isinstance(final_pred, pred_type), (
|
||||
f"Type mismatch: 'final_pred' is of type {type(final_pred)}, but expected {pred_type} "
|
||||
)
|
||||
|
||||
# Check shape
|
||||
if isinstance(final_pred, (list, np.ndarray, pd.DataFrame, torch.Tensor, tf.Tensor)):
|
||||
assert get_length(final_pred) == get_length(test_X), (
|
||||
f"Wrong output sample size: get_length(final_pred)={get_length(final_pred)} "
|
||||
f"vs. get_length(test_X)={get_length(test_X)}"
|
||||
)
|
||||
|
||||
# check scores.csv
|
||||
assert Path("scores.csv").exists(), "scores.csv is not generated"
|
||||
score_df = pd.read_csv("scores.csv", index_col=0)
|
||||
model_set_in_scores = set(score_df.index)
|
||||
|
||||
assert model_set_in_scores == set({{model_names}}).union({"ensemble"}), (
|
||||
f"The scores dataframe does not contain the correct model names as index.\ncorrect model names are: {{model_names}} + ['ensemble']\nscore_df is:\n{score_df}"
|
||||
)
|
||||
assert score_df.index.is_unique, "The scores dataframe has duplicate model names."
|
||||
assert score_df.columns.tolist() == ["{{metric_name}}"], f"The column names of the scores dataframe should be ['{{metric_name}}'], but is '{score_df.columns.tolist()}'"
|
||||
|
||||
# Check for NaN values in score_df
|
||||
assert not score_df.isnull().values.any(), (
|
||||
f"The scores dataframe contains NaN values at the following locations:\n{score_df[score_df.isnull().any(axis=1)]}"
|
||||
)
|
||||
|
||||
|
||||
print("Ensemble test end.")
|
||||
@@ -0,0 +1,13 @@
|
||||
import pickle
|
||||
import site
|
||||
import traceback
|
||||
from pathlib import Path
|
||||
from typing import Dict, Optional
|
||||
|
||||
from rdagent.components.coder.CoSTEER.task import CoSTEERTask
|
||||
from rdagent.core.utils import cache_with_pickle
|
||||
|
||||
|
||||
# Because we use isinstance to distinguish between different types of tasks, we need to use sub classes to represent different types of tasks
|
||||
class EnsembleTask(CoSTEERTask):
|
||||
pass
|
||||
@@ -0,0 +1,124 @@
|
||||
ensemble_coder:
|
||||
system: |-
|
||||
You are a world-class data scientist and machine learning engineer with deep expertise in statistics, mathematics, and computer science.
|
||||
Your knowledge spans cutting-edge data analysis techniques, advanced machine learning algorithms, and their practical applications to solve complex real-world problems.
|
||||
|
||||
## Task Description
|
||||
Currently, you are working on model ensemble implementation. Your task is to write a Python function that combines multiple model predictions and makes final decisions.
|
||||
|
||||
Your specific task as follows:
|
||||
{{ task_desc }}
|
||||
|
||||
## Competition Information for This Task
|
||||
{{ competition_info }}
|
||||
|
||||
{% if queried_similar_successful_knowledge|length != 0 or queried_former_failed_knowledge|length != 0 %}
|
||||
## Relevant Information for This Task
|
||||
{% endif %}
|
||||
|
||||
{% if queried_similar_successful_knowledge|length != 0 %}
|
||||
--------- Successful Implementations for Similar Models ---------
|
||||
====={% for similar_successful_knowledge in queried_similar_successful_knowledge %} Model {{ loop.index }}:=====
|
||||
{{ similar_successful_knowledge.target_task.get_task_information() }}
|
||||
=====Code:=====
|
||||
{{ similar_successful_knowledge.implementation.file_dict["ensemble.py"] }}
|
||||
{% endfor %}
|
||||
{% endif %}
|
||||
|
||||
{% if queried_former_failed_knowledge|length != 0 %}
|
||||
--------- Previous Failed Attempts ---------
|
||||
{% for former_failed_knowledge in queried_former_failed_knowledge %} Attempt {{ loop.index }}:
|
||||
=====Code:=====
|
||||
{{ former_failed_knowledge.implementation.file_dict["ensemble.py"] }}
|
||||
=====Feedback:=====
|
||||
{{ former_failed_knowledge.feedback }}
|
||||
{% endfor %}
|
||||
{% endif %}
|
||||
|
||||
## Guidelines
|
||||
1. The function's code is associated with several other functions including a data loader, feature engineering, and model training. all codes are as follows:
|
||||
{{ all_code }}
|
||||
2. You should avoid using logging module to output information in your generated code, and instead use the print() function.
|
||||
{% include "scenarios.data_science.share:guidelines.coding" %}
|
||||
|
||||
## Output Format
|
||||
{% if out_spec %}
|
||||
{{ out_spec }}
|
||||
{% else %}
|
||||
Please response the code in the following json format. Here is an example structure for the JSON output:
|
||||
{
|
||||
"code": "The Python code as a string."
|
||||
}
|
||||
{% endif %}
|
||||
|
||||
user: |-
|
||||
--------- Code Specification ---------
|
||||
{{ code_spec }}
|
||||
|
||||
{% if latest_code %}
|
||||
--------- Former code ---------
|
||||
{{ latest_code }}
|
||||
{% if latest_code_feedback is not none %}
|
||||
--------- Feedback to former code ---------
|
||||
{{ latest_code_feedback }}
|
||||
{% endif %}
|
||||
The former code contains errors. You should correct the code based on the provided information, ensuring you do not repeat the same mistakes.
|
||||
{% endif %}
|
||||
|
||||
|
||||
ensemble_eval:
|
||||
system: |-
|
||||
You are a data scientist responsible for evaluating ensemble implementation code generation.
|
||||
|
||||
## Task Description
|
||||
{{ task_desc }}
|
||||
|
||||
## Ensemble Code
|
||||
```python
|
||||
{{ code }}
|
||||
```
|
||||
|
||||
## Testing Process
|
||||
The ensemble code is tested using the following script:
|
||||
```python
|
||||
{{ test_code }}
|
||||
```
|
||||
You will analyze the execution results based on the test output provided.
|
||||
|
||||
{% if workflow_stdout is not none %}
|
||||
### Whole Workflow Consideration
|
||||
The ensemble code is part of the whole workflow. The user has executed the entire pipeline and provided additional stdout.
|
||||
|
||||
**Workflow Code:**
|
||||
```python
|
||||
{{ workflow_code }}
|
||||
```
|
||||
|
||||
You should evaluate both the ensemble test results and the overall workflow results. **Approve the code only if both tests pass.**
|
||||
{% endif %}
|
||||
|
||||
The metric used for scoring the predictions:
|
||||
**{{ metric_name }}**
|
||||
|
||||
## Evaluation Criteria
|
||||
- You will be given the standard output (`stdout`) from the ensemble test and, if applicable, the workflow test.
|
||||
- Code should have no try-except blocks because they can hide errors.
|
||||
- Check whether the code implement the scoring process using the given metric.
|
||||
- The stdout includes the local variable values from the ensemble code execution. Check whether the validation score is calculated correctly.
|
||||
|
||||
Please respond with your feedback in the following JSON format and order
|
||||
```json
|
||||
{
|
||||
"execution": "Describe how well the ensemble executed, including any errors or issues encountered. Append all error messages and full traceback details without summarizing or omitting any information.",
|
||||
"return_checking": "Detail the checks performed on the ensemble results, including shape and value validation.",
|
||||
"code": "Assess code quality, readability, and adherence to specifications.",
|
||||
"final_decision": <true/false>
|
||||
}
|
||||
```
|
||||
user: |-
|
||||
--------- Ensemble test stdout ---------
|
||||
{{ stdout }}
|
||||
{% if workflow_stdout is not none %}
|
||||
--------- Whole workflow test stdout ---------
|
||||
{{ workflow_stdout }}
|
||||
{% endif %}
|
||||
@@ -0,0 +1,58 @@
|
||||
"""
|
||||
Helper functions for testing the ensemble coder(CoSTEER-based) component.
|
||||
"""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from rdagent.components.coder.data_science.ensemble import EnsembleCoSTEER
|
||||
from rdagent.components.coder.data_science.ensemble.exp import EnsembleTask
|
||||
from rdagent.scenarios.data_science.experiment.experiment import DSExperiment
|
||||
from rdagent.scenarios.data_science.scen import KaggleScen
|
||||
|
||||
# Add the competition folder to path
|
||||
COMPETITION_PATH = (
|
||||
Path(__file__).parent.parent.parent.parent.parent
|
||||
/ "scenarios"
|
||||
/ "kaggle"
|
||||
/ "tpl_ex"
|
||||
/ "aerial-cactus-identification"
|
||||
)
|
||||
sys.path.append(str(COMPETITION_PATH))
|
||||
|
||||
EnsembleExperiment = DSExperiment
|
||||
|
||||
|
||||
def load_ensemble_spec():
|
||||
spec_path = COMPETITION_PATH / "spec" / "ensemble.md"
|
||||
with open(spec_path, "r") as f:
|
||||
return f.read()
|
||||
|
||||
|
||||
def develop_one_competition(competition: str):
|
||||
# Initialize scenario and coder
|
||||
scen = KaggleScen(competition=competition)
|
||||
ensemble_coder = EnsembleCoSTEER(scen)
|
||||
# Load ensemble specification
|
||||
ensemble_spec = load_ensemble_spec()
|
||||
|
||||
# Create the ensemble task with actual data context and specification
|
||||
task = EnsembleTask(
|
||||
name="EnsembleTask",
|
||||
description="""
|
||||
Implement ensemble and decision making for model predictions.
|
||||
""",
|
||||
)
|
||||
|
||||
exp = EnsembleExperiment(pending_tasks_list=[task])
|
||||
|
||||
# Injecting the corresponding specification
|
||||
exp.experiment_workspace.inject_files(**{"spec/ensemble.md": ensemble_spec})
|
||||
|
||||
# Develop the experiment
|
||||
exp = ensemble_coder.develop(exp)
|
||||
return exp
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
develop_one_competition("aerial-cactus-identification")
|
||||
@@ -0,0 +1,142 @@
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Dict
|
||||
|
||||
from rdagent.app.data_science.conf import DS_RD_SETTING
|
||||
from rdagent.components.coder.CoSTEER import CoSTEER
|
||||
from rdagent.components.coder.CoSTEER.evaluators import (
|
||||
CoSTEERMultiEvaluator,
|
||||
CoSTEERSingleFeedback,
|
||||
)
|
||||
from rdagent.components.coder.CoSTEER.evolving_strategy import (
|
||||
MultiProcessEvolvingStrategy,
|
||||
)
|
||||
from rdagent.components.coder.CoSTEER.knowledge_management import (
|
||||
CoSTEERQueriedKnowledge,
|
||||
)
|
||||
from rdagent.components.coder.data_science.conf import DSCoderCoSTEERSettings
|
||||
from rdagent.components.coder.data_science.feature.eval import FeatureCoSTEEREvaluator
|
||||
from rdagent.components.coder.data_science.feature.exp import FeatureTask
|
||||
from rdagent.core.exception import CoderError
|
||||
from rdagent.core.experiment import FBWorkspace
|
||||
from rdagent.core.scenario import Scenario
|
||||
from rdagent.oai.llm_utils import APIBackend
|
||||
from rdagent.utils.agent.ret import PythonAgentOut
|
||||
from rdagent.utils.agent.tpl import T
|
||||
|
||||
DIRNAME = Path(__file__).absolute().resolve().parent
|
||||
|
||||
|
||||
class FeatureMultiProcessEvolvingStrategy(MultiProcessEvolvingStrategy):
|
||||
def implement_one_task(
|
||||
self,
|
||||
target_task: FeatureTask,
|
||||
queried_knowledge: CoSTEERQueriedKnowledge | None = None,
|
||||
workspace: FBWorkspace | None = None,
|
||||
prev_task_feedback: CoSTEERSingleFeedback | None = None,
|
||||
) -> dict[str, str]:
|
||||
# return a workspace with "load_data.py", "spec/load_data.md" inside
|
||||
# assign the implemented code to the new workspace.
|
||||
feature_information_str = target_task.get_task_information()
|
||||
|
||||
# 1. query
|
||||
queried_similar_successful_knowledge = (
|
||||
queried_knowledge.task_to_similar_task_successful_knowledge[feature_information_str]
|
||||
if queried_knowledge is not None
|
||||
else []
|
||||
)
|
||||
queried_former_failed_knowledge = (
|
||||
queried_knowledge.task_to_former_failed_traces[feature_information_str]
|
||||
if queried_knowledge is not None
|
||||
else []
|
||||
)
|
||||
queried_former_failed_knowledge = (
|
||||
[
|
||||
knowledge
|
||||
for knowledge in queried_former_failed_knowledge[0]
|
||||
if knowledge.implementation.file_dict.get("feature.py") != workspace.file_dict.get("feature.py")
|
||||
],
|
||||
queried_former_failed_knowledge[1],
|
||||
)
|
||||
|
||||
# 2. code
|
||||
system_prompt = T(".prompts:feature_coder.system").r(
|
||||
competition_info=self.scen.get_scenario_all_desc(eda_output=workspace.file_dict.get("EDA.md", None)),
|
||||
task_desc=feature_information_str,
|
||||
data_loader_code=workspace.file_dict.get("load_data.py"),
|
||||
queried_similar_successful_knowledge=queried_similar_successful_knowledge,
|
||||
queried_former_failed_knowledge=queried_former_failed_knowledge[0],
|
||||
out_spec=PythonAgentOut.get_spec(),
|
||||
)
|
||||
code_spec = (
|
||||
workspace.file_dict["spec/feature.md"]
|
||||
if DS_RD_SETTING.spec_enabled
|
||||
else T("scenarios.data_science.share:component_spec.general").r(
|
||||
spec=T("scenarios.data_science.share:component_spec.FeatureEng").r(),
|
||||
test_code=(DIRNAME / "eval_tests" / "feature_test.txt").read_text(),
|
||||
)
|
||||
)
|
||||
user_prompt = T(".prompts:feature_coder.user").r(
|
||||
code_spec=code_spec,
|
||||
latest_code=workspace.file_dict.get("feature.py"),
|
||||
latest_code_feedback=prev_task_feedback,
|
||||
)
|
||||
|
||||
for _ in range(5):
|
||||
feature_code = PythonAgentOut.extract_output(
|
||||
APIBackend().build_messages_and_create_chat_completion(
|
||||
user_prompt=user_prompt,
|
||||
system_prompt=system_prompt,
|
||||
)
|
||||
)
|
||||
if feature_code != workspace.file_dict.get("feature.py"):
|
||||
break
|
||||
else:
|
||||
user_prompt = user_prompt + "\nPlease avoid generating same code to former code!"
|
||||
else:
|
||||
raise CoderError("Failed to generate a new feature code.")
|
||||
|
||||
return {
|
||||
"feature.py": feature_code,
|
||||
}
|
||||
|
||||
def assign_code_list_to_evo(self, code_list: list[dict[str, str]], evo):
|
||||
"""
|
||||
Assign the code list to the evolving item.
|
||||
|
||||
The code list is aligned with the evolving item's sub-tasks.
|
||||
If a task is not implemented, put a None in the list.
|
||||
"""
|
||||
for index in range(len(evo.sub_tasks)):
|
||||
if code_list[index] is None:
|
||||
continue
|
||||
if evo.sub_workspace_list[index] is None:
|
||||
# evo.sub_workspace_list[index] = FBWorkspace(target_task=evo.sub_tasks[index])
|
||||
evo.sub_workspace_list[index] = evo.experiment_workspace
|
||||
evo.sub_workspace_list[index].inject_files(**code_list[index])
|
||||
return evo
|
||||
|
||||
|
||||
class FeatureCoSTEER(CoSTEER):
|
||||
def __init__(
|
||||
self,
|
||||
scen: Scenario,
|
||||
*args,
|
||||
**kwargs,
|
||||
) -> None:
|
||||
settings = DSCoderCoSTEERSettings()
|
||||
eva = CoSTEERMultiEvaluator(
|
||||
FeatureCoSTEEREvaluator(scen=scen), scen=scen
|
||||
) # Please specify whether you agree running your eva in parallel or not
|
||||
es = FeatureMultiProcessEvolvingStrategy(scen=scen, settings=settings)
|
||||
|
||||
super().__init__(
|
||||
*args,
|
||||
settings=settings,
|
||||
eva=eva,
|
||||
es=es,
|
||||
evolving_version=2,
|
||||
scen=scen,
|
||||
max_loop=DS_RD_SETTING.coder_max_loop,
|
||||
**kwargs,
|
||||
)
|
||||
@@ -0,0 +1,81 @@
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
from rdagent.app.data_science.conf import DS_RD_SETTING
|
||||
from rdagent.components.coder.CoSTEER.evaluators import (
|
||||
CoSTEEREvaluator,
|
||||
CoSTEERSingleFeedback,
|
||||
)
|
||||
from rdagent.components.coder.data_science.conf import get_ds_env
|
||||
from rdagent.components.coder.data_science.utils import remove_eda_part
|
||||
from rdagent.core.evolving_framework import QueriedKnowledge
|
||||
from rdagent.core.experiment import FBWorkspace, Task
|
||||
from rdagent.utils.agent.tpl import T
|
||||
from rdagent.utils.agent.workflow import build_cls_from_json_with_retry
|
||||
from rdagent.utils.fmt import shrink_text
|
||||
|
||||
DIRNAME = Path(__file__).absolute().resolve().parent
|
||||
|
||||
FeatureEvalFeedback = CoSTEERSingleFeedback
|
||||
|
||||
|
||||
class FeatureCoSTEEREvaluator(CoSTEEREvaluator):
|
||||
def evaluate(
|
||||
self,
|
||||
target_task: Task,
|
||||
implementation: FBWorkspace,
|
||||
gt_implementation: FBWorkspace,
|
||||
queried_knowledge: QueriedKnowledge = None,
|
||||
**kwargs,
|
||||
) -> FeatureEvalFeedback:
|
||||
target_task_information = target_task.get_task_information()
|
||||
if (
|
||||
queried_knowledge is not None
|
||||
and target_task_information in queried_knowledge.success_task_to_knowledge_dict
|
||||
):
|
||||
return queried_knowledge.success_task_to_knowledge_dict[target_task_information].feedback
|
||||
elif queried_knowledge is not None and target_task_information in queried_knowledge.failed_task_info_set:
|
||||
return FeatureEvalFeedback(
|
||||
execution="This task has failed too many times, skip implementation.",
|
||||
return_checking="This task has failed too many times, skip implementation.",
|
||||
code="This task has failed too many times, skip implementation.",
|
||||
final_decision=False,
|
||||
)
|
||||
|
||||
env = get_ds_env(extra_volumes={self.scen.debug_path: T("scenarios.data_science.share:scen.input_path").r()})
|
||||
|
||||
# TODO: do we need to clean the generated temporary content?
|
||||
fname = "test/feature_test.py"
|
||||
test_code = (DIRNAME / "eval_tests" / "feature_test.txt").read_text()
|
||||
implementation.inject_files(**{fname: test_code})
|
||||
|
||||
stdout, ret_code = implementation.execute_ret_code(env=env, entry=f"python {fname}")
|
||||
|
||||
if "main.py" in implementation.file_dict and ret_code == 0:
|
||||
workflow_stdout = implementation.execute(env=env, entry="python main.py")
|
||||
workflow_stdout = remove_eda_part(workflow_stdout)
|
||||
else:
|
||||
workflow_stdout = None
|
||||
|
||||
system_prompt = T(".prompts:feature_eval.system").r(
|
||||
task_desc=target_task.get_task_information(),
|
||||
test_code=test_code,
|
||||
code=implementation.file_dict["feature.py"],
|
||||
workflow_stdout=workflow_stdout,
|
||||
workflow_code=implementation.all_codes,
|
||||
)
|
||||
user_prompt = T(".prompts:feature_eval.user").r(
|
||||
stdout=shrink_text(stdout),
|
||||
workflow_stdout=workflow_stdout,
|
||||
)
|
||||
|
||||
fb = build_cls_from_json_with_retry(
|
||||
FeatureEvalFeedback,
|
||||
system_prompt=system_prompt,
|
||||
user_prompt=user_prompt,
|
||||
init_kwargs_update_func=FeatureEvalFeedback.val_and_update_init_dict,
|
||||
)
|
||||
fb.final_decision = fb.final_decision and ret_code == 0
|
||||
|
||||
return fb
|
||||
@@ -0,0 +1,114 @@
|
||||
"""
|
||||
Tests for `feat_eng` in feature.py
|
||||
"""
|
||||
|
||||
|
||||
from copy import deepcopy
|
||||
import sys
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from feature import feat_eng
|
||||
from load_data import load_data
|
||||
import reprlib
|
||||
aRepr = reprlib.Repr()
|
||||
aRepr.maxother=300
|
||||
|
||||
X, y, X_test, test_ids = load_data()
|
||||
print("X:", aRepr.repr(X))
|
||||
print("y:", aRepr.repr(y))
|
||||
print("X_test:", aRepr.repr(X_test))
|
||||
print("test_ids", aRepr.repr(test_ids))
|
||||
|
||||
print(f"X.shape: {X.shape}" if hasattr(X, 'shape') else f"X length: {len(X)}")
|
||||
print(f"y.shape: {y.shape}" if hasattr(y, 'shape') else f"y length: {len(y)}")
|
||||
print(f"X_test.shape: {X_test.shape}" if hasattr(X_test, 'shape') else f"X_test length: {len(X_test)}")
|
||||
print(f"test_ids length: {len(test_ids)}")
|
||||
|
||||
X_loaded = deepcopy(X)
|
||||
y_loaded = deepcopy(y)
|
||||
X_test_loaded = deepcopy(X_test)
|
||||
|
||||
import sys
|
||||
import reprlib
|
||||
from joblib.memory import MemorizedFunc
|
||||
|
||||
|
||||
def get_original_code(func):
|
||||
if isinstance(func, MemorizedFunc):
|
||||
return func.func.__code__
|
||||
return func.__code__
|
||||
|
||||
|
||||
def debug_info_print(func):
|
||||
def wrapper(*args, **kwargs):
|
||||
original_code = get_original_code(func)
|
||||
def local_trace(frame, event, arg):
|
||||
if event == "return" and frame.f_code == original_code:
|
||||
print("\n" + "="*20 + "Running feat_eng code, local variable values:" + "="*20)
|
||||
for k, v in frame.f_locals.items():
|
||||
printed = aRepr.repr(v)
|
||||
print(f"{k}:\n {printed}")
|
||||
print("="*20 + "Local variable values end" + "="*20)
|
||||
return local_trace
|
||||
|
||||
sys.settrace(local_trace)
|
||||
try:
|
||||
return func(*args, **kwargs)
|
||||
finally:
|
||||
sys.settrace(None)
|
||||
return wrapper
|
||||
X, y, X_test = debug_info_print(feat_eng)(X, y, X_test)
|
||||
|
||||
|
||||
def get_length(data):
|
||||
return data.shape[0] if hasattr(data, 'shape') else len(data)
|
||||
|
||||
|
||||
def get_width(data):
|
||||
return 1 if isinstance(data, list) else data.shape[1:]
|
||||
|
||||
|
||||
def get_column_list(data):
|
||||
return data.columns.tolist() if isinstance(data, pd.DataFrame) else None
|
||||
|
||||
|
||||
assert X is not None, "The feature engineering function returned None for X."
|
||||
assert y is not None, "The feature engineering function returned None for y."
|
||||
assert X_test is not None, "The feature engineering function returned None for X_test."
|
||||
|
||||
assert get_length(X_test) == get_length(
|
||||
test_ids
|
||||
), f"Mismatch in length of test images and test IDs: X_test ({get_length(X_test)}) and test_ids ({get_length(test_ids)})"
|
||||
assert get_length(X) == get_length(
|
||||
y
|
||||
), f"Mismatch in length of training images and labels: X ({get_length(X)}) and y ({get_length(y)})"
|
||||
|
||||
assert get_length(X) != 0, f"Training data is empty."
|
||||
assert get_length(y) != 0, f"Training labels are empty."
|
||||
assert get_length(X_test) != 0, f"Test data is empty."
|
||||
|
||||
assert get_width(X) == get_width(
|
||||
X_test
|
||||
), "Mismatch in width of training and test data. Width means the number of features."
|
||||
|
||||
if isinstance(X, pd.DataFrame) and isinstance(X_test, pd.DataFrame):
|
||||
assert get_column_list(X) == get_column_list(X_test), "Mismatch in column names of training and test data."
|
||||
|
||||
if isinstance(X, pd.DataFrame):
|
||||
def normalize_dtype(dtype):
|
||||
return "numeric" if np.issubdtype(dtype, np.number) else str(dtype)
|
||||
|
||||
X_dtypes_unique_sorted = sorted(set(normalize_dtype(dt) for dt in X.dtypes.unique()))
|
||||
X_loaded_dtypes_unique_sorted = sorted(set(normalize_dtype(dt) for dt in X_loaded.dtypes.unique()))
|
||||
|
||||
X_dtypes_unique_sorted_new = [
|
||||
dt for dt in X_dtypes_unique_sorted if dt not in X_loaded_dtypes_unique_sorted and dt != "object"
|
||||
]
|
||||
assert (
|
||||
np.dtypes.ObjectDType in X_loaded_dtypes_unique_sorted or len(X_dtypes_unique_sorted_new) == 0
|
||||
), f"feature engineering has produced new data types which is not allowed, data loader data types are {X_loaded_dtypes_unique_sorted} and feature engineering data types are {X_dtypes_unique_sorted}"
|
||||
|
||||
|
||||
print(
|
||||
"Feature Engineering test passed successfully. All checks including length, width, and data types have been validated."
|
||||
)
|
||||
@@ -0,0 +1,13 @@
|
||||
import pickle
|
||||
import site
|
||||
import traceback
|
||||
from pathlib import Path
|
||||
from typing import Dict, Optional
|
||||
|
||||
from rdagent.components.coder.CoSTEER.task import CoSTEERTask
|
||||
from rdagent.core.utils import cache_with_pickle
|
||||
|
||||
|
||||
# Because we use isinstance to distinguish between different types of tasks, we need to use sub classes to represent different types of tasks
|
||||
class FeatureTask(CoSTEERTask):
|
||||
pass
|
||||
@@ -0,0 +1,131 @@
|
||||
feature_coder:
|
||||
system: |-
|
||||
You are a world-class data scientist and machine learning engineer with deep expertise in statistics, mathematics, and computer science.
|
||||
Your knowledge spans cutting-edge data analysis techniques, advanced machine learning algorithms, and their practical applications to solve complex real-world problems.
|
||||
|
||||
## Task Description
|
||||
{{ task_desc }}
|
||||
|
||||
## Competition Information for This Task
|
||||
{{ competition_info }}
|
||||
|
||||
{% if queried_similar_successful_knowledge|length != 0 or queried_former_failed_knowledge|length != 0 %}
|
||||
## Relevant Information for This Task
|
||||
{% endif %}
|
||||
|
||||
{% if queried_similar_successful_knowledge|length != 0 %}
|
||||
--------- Successful Implementations for Similar Models ---------
|
||||
====={% for similar_successful_knowledge in queried_similar_successful_knowledge %} Model {{ loop.index }}:=====
|
||||
{{ similar_successful_knowledge.target_task.get_task_information() }}
|
||||
=====Code:=====
|
||||
{{ similar_successful_knowledge.implementation.file_dict["feature.py"] }}
|
||||
{% endfor %}
|
||||
{% endif %}
|
||||
|
||||
{% if queried_former_failed_knowledge|length != 0 %}
|
||||
--------- Previous Failed Attempts ---------
|
||||
{% for former_failed_knowledge in queried_former_failed_knowledge %} Attempt {{ loop.index }}:
|
||||
=====Code:=====
|
||||
{{ former_failed_knowledge.implementation.file_dict["feature.py"] }}
|
||||
=====Feedback:=====
|
||||
{{ former_failed_knowledge.feedback }}
|
||||
{% endfor %}
|
||||
{% endif %}
|
||||
|
||||
## Guidelines
|
||||
1. If feature engineering is unnecessary or should be combined with model training, you may skip this step.
|
||||
2. Be cautious of any column drop in the code. Dropping a column easily without any more attempts, it may not be a good practice.
|
||||
3. The function input is the output of the following data loader:
|
||||
```python
|
||||
{{ data_loader_code }}
|
||||
```
|
||||
4. **Additional Guidance:**
|
||||
- If a previous attempt exists, improve upon it without repeating mistakes.
|
||||
- If errors indicate a missing file, find a way to download it or implement an alternative solution.
|
||||
- You should avoid using logging module to output information in your generated code, and instead use the print() function.
|
||||
5. You should use the following cache decorator to cache the results of the function:
|
||||
```python
|
||||
from joblib import Memory
|
||||
memory = Memory(location='{% include "scenarios.data_science.share:scen.cache_path" %}', verbose=0)
|
||||
@memory.cache```
|
||||
6. Coding tricks:
|
||||
- If the input consists of a batch of file paths and you need to modify the file contents to complete your feature engineering task, you can accomplish your feature engineering task by modifying these files and creating new files in a subfolder within "{% include "scenarios.data_science.share:scen.cache_path" %}" (this path is persistent, otherwise you may lose your created file). Then the new file paths are returned.
|
||||
|
||||
{% include "scenarios.data_science.share:guidelines.coding" %}
|
||||
|
||||
## Output Format
|
||||
{% if out_spec %}
|
||||
{{ out_spec }}
|
||||
{% else %}
|
||||
Please response the code in the following json format. Here is an example structure for the JSON output:
|
||||
{
|
||||
"code": "The Python code as a string."
|
||||
}
|
||||
{% endif %}
|
||||
|
||||
user: |-
|
||||
--------- Code Specification ---------
|
||||
{{ code_spec }}
|
||||
|
||||
{% if latest_code %}
|
||||
--------- Former code ---------
|
||||
{{ latest_code }}
|
||||
{% if latest_code_feedback is not none %}
|
||||
--------- Feedback to former code ---------
|
||||
{{ latest_code_feedback }}
|
||||
{% endif %}
|
||||
The former code contains errors. You should correct the code based on the provided information, ensuring you do not repeat the same mistakes.
|
||||
{% endif %}
|
||||
|
||||
|
||||
feature_eval:
|
||||
system: |-
|
||||
You are a data scientist responsible for evaluating feature engineering code generation.
|
||||
|
||||
## Task Description
|
||||
{{ task_desc }}
|
||||
|
||||
## Feature Engineering Code
|
||||
```python
|
||||
{{ code }}
|
||||
```
|
||||
|
||||
## Testing Process
|
||||
The feature engineering code is tested using the following script:
|
||||
```python
|
||||
{{ test_code }}
|
||||
```
|
||||
You will analyze the execution results based on the test output provided.
|
||||
|
||||
{% if workflow_stdout is not none %}
|
||||
### Whole Workflow Consideration
|
||||
The feature engineering code is part of the whole workflow. The user has executed the entire pipeline and provided additional stdout.
|
||||
|
||||
**Workflow Code:**
|
||||
```python
|
||||
{{ workflow_code }}
|
||||
```
|
||||
|
||||
You should evaluate both the feature engineering test results and the overall workflow results. **Approve the code only if both tests pass.**
|
||||
{% endif %}
|
||||
|
||||
## Evaluation Criteria
|
||||
You will be given the standard output (`stdout`) from the feature engineering test and, if applicable, the workflow test.
|
||||
|
||||
Please respond with your feedback in the following JSON format and order
|
||||
```json
|
||||
{
|
||||
"execution": "Describe how well the feature engineering executed, including any errors or issues encountered. Append all error messages and full traceback details without summarizing or omitting any information.",
|
||||
"return_checking": "Evaluate the correctness and integrity of processed data, checking for missing values, incorrect transformations, and data consistency.",
|
||||
"code": "Assess code quality, readability, and adherence to specifications. Consider efficiency, including whether the code utilizes multi-threading or GPU acceleration for optimization.",
|
||||
"final_decision": <true/false>
|
||||
}
|
||||
```
|
||||
|
||||
user: |-
|
||||
--------- Feature engineering test stdout ---------
|
||||
{{ stdout }}
|
||||
{% if workflow_stdout is not none %}
|
||||
--------- Whole workflow test stdout ---------
|
||||
{{ workflow_stdout }}
|
||||
{% endif %}
|
||||
@@ -0,0 +1,37 @@
|
||||
"""
|
||||
Helper functions for testing the feature coder(CoSTEER-based) component.
|
||||
- Does the developer loop work correctly
|
||||
|
||||
It is NOT:
|
||||
- it is not interface unittest(i.e. workspace evaluator in the CoSTEER Loop)
|
||||
"""
|
||||
|
||||
from rdagent.components.coder.data_science.feature import FeatureCoSTEER
|
||||
from rdagent.components.coder.data_science.feature.exp import FeatureTask
|
||||
from rdagent.scenarios.data_science.experiment.experiment import DSExperiment
|
||||
from rdagent.scenarios.data_science.scen import KaggleScen
|
||||
|
||||
|
||||
def develop_one_competition(competition: str): # -> experiment
|
||||
scen = KaggleScen(competition=competition)
|
||||
feature_coder = FeatureCoSTEER(scen)
|
||||
|
||||
with open("./rdagent/scenarios/kaggle/tpl_ex/aerial-cactus-identification/spec/feature.md", "r") as file:
|
||||
feat_spec = file.read()
|
||||
|
||||
# Create the experiment
|
||||
ft = FeatureTask(name="FeatureTask", description=scen.get_competition_full_desc())
|
||||
exp = DSExperiment(
|
||||
sub_tasks=[ft],
|
||||
)
|
||||
|
||||
with open("./rdagent/scenarios/kaggle/tpl_ex/aerial-cactus-identification/load_data.py", "r") as file:
|
||||
load_data_code = file.read()
|
||||
exp.experiment_workspace.inject_files(**{"load_data.py": load_data_code, "spec/feature.md": feat_spec})
|
||||
|
||||
# Develop the experiment
|
||||
exp = feature_coder.develop(exp)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
develop_one_competition("aerial-cactus-identification")
|
||||
@@ -0,0 +1,174 @@
|
||||
from pathlib import Path
|
||||
from typing import Dict
|
||||
|
||||
from rdagent.app.data_science.conf import DS_RD_SETTING
|
||||
from rdagent.components.coder.CoSTEER import CoSTEER
|
||||
from rdagent.components.coder.CoSTEER.evaluators import (
|
||||
CoSTEERMultiEvaluator,
|
||||
CoSTEERSingleFeedback,
|
||||
)
|
||||
from rdagent.components.coder.CoSTEER.evolving_strategy import (
|
||||
MultiProcessEvolvingStrategy,
|
||||
)
|
||||
from rdagent.components.coder.CoSTEER.knowledge_management import (
|
||||
CoSTEERQueriedKnowledge,
|
||||
)
|
||||
from rdagent.components.coder.data_science.conf import DSCoderCoSTEERSettings
|
||||
from rdagent.components.coder.data_science.model.eval import (
|
||||
ModelGeneralCaseSpecEvaluator,
|
||||
)
|
||||
from rdagent.components.coder.data_science.model.exp import ModelTask
|
||||
from rdagent.core.exception import CoderError
|
||||
from rdagent.core.experiment import FBWorkspace
|
||||
from rdagent.core.scenario import Scenario
|
||||
from rdagent.oai.llm_utils import APIBackend
|
||||
from rdagent.utils.agent.ret import PythonBatchEditOut
|
||||
from rdagent.utils.agent.tpl import T
|
||||
|
||||
DIRNAME = Path(__file__).absolute().resolve().parent
|
||||
|
||||
|
||||
class ModelMultiProcessEvolvingStrategy(MultiProcessEvolvingStrategy):
|
||||
def implement_one_task(
|
||||
self,
|
||||
target_task: ModelTask,
|
||||
queried_knowledge: CoSTEERQueriedKnowledge | None = None,
|
||||
workspace: FBWorkspace | None = None,
|
||||
prev_task_feedback: CoSTEERSingleFeedback | None = None,
|
||||
) -> dict[str, str]:
|
||||
model_information_str = target_task.get_task_information()
|
||||
|
||||
# 1. query
|
||||
queried_similar_successful_knowledge = (
|
||||
queried_knowledge.task_to_similar_task_successful_knowledge[model_information_str]
|
||||
if queried_knowledge is not None
|
||||
else []
|
||||
)
|
||||
queried_former_failed_knowledge = (
|
||||
queried_knowledge.task_to_former_failed_traces[model_information_str]
|
||||
if queried_knowledge is not None
|
||||
else []
|
||||
)
|
||||
queried_former_failed_knowledge = (
|
||||
[
|
||||
knowledge
|
||||
for knowledge in queried_former_failed_knowledge[0]
|
||||
if knowledge.implementation.file_dict.get(f"{target_task.name}.py")
|
||||
!= workspace.file_dict.get(f"{target_task.name}.py")
|
||||
],
|
||||
queried_former_failed_knowledge[1],
|
||||
)
|
||||
|
||||
# 2. code
|
||||
system_prompt = T(".prompts:model_coder.system").r(
|
||||
task_desc=model_information_str,
|
||||
competition_info=self.scen.get_scenario_all_desc(eda_output=workspace.file_dict.get("EDA.md", None)),
|
||||
data_loader_code=workspace.file_dict.get("load_data.py"),
|
||||
feature_code=workspace.file_dict["feature.py"],
|
||||
queried_similar_successful_knowledge=queried_similar_successful_knowledge,
|
||||
queried_former_failed_knowledge=queried_former_failed_knowledge[0],
|
||||
out_spec=PythonBatchEditOut.get_spec(),
|
||||
)
|
||||
# user_prompt = T(".prompts:model_coder.user").r(
|
||||
# model_spec=workspace.file_dict["spec/model.md"],
|
||||
# feature_code=workspace.file_dict["feature.py"],
|
||||
# latest_code=workspace.file_dict.get(f"{target_task.name}.py", None),
|
||||
# )
|
||||
# We want to use a simpler way to
|
||||
code_spec = (
|
||||
workspace.file_dict["spec/model.md"]
|
||||
if DS_RD_SETTING.spec_enabled
|
||||
else T("scenarios.data_science.share:component_spec.general").r(
|
||||
spec=T("scenarios.data_science.share:component_spec.Model").r(),
|
||||
test_code=(DIRNAME / "eval_tests" / "model_test.txt").read_text().replace("model01", target_task.name),
|
||||
)
|
||||
)
|
||||
user_prompt = T(".prompts:model_coder.user_general").r(
|
||||
code_spec=code_spec,
|
||||
latest_model_code=workspace.get_codes(
|
||||
r"^model_(?!test)\w+\.py$"
|
||||
), # TODO: If we have high failure rate here, we should clean this step with less information.
|
||||
latest_code_feedback=prev_task_feedback,
|
||||
)
|
||||
|
||||
for _ in range(5):
|
||||
batch_edit = PythonBatchEditOut.extract_output(
|
||||
APIBackend().build_messages_and_create_chat_completion(
|
||||
user_prompt=user_prompt,
|
||||
system_prompt=system_prompt,
|
||||
)
|
||||
)
|
||||
|
||||
if not all(i.startswith("model_") for i in batch_edit.keys()):
|
||||
user_prompt += "\nYou should only update model codes!"
|
||||
continue
|
||||
|
||||
# 3. post process to align file name to the task name
|
||||
# we assumpt batch_edit only contains one model file update.
|
||||
batch_edit = {
|
||||
(f"{target_task.name}.py" if value != "__DEL__" and key != f"{target_task.name}.py" else key): value
|
||||
for key, value in batch_edit.items()
|
||||
}
|
||||
|
||||
user_prompt = user_prompt + "\nPlease avoid generating same code to former code!"
|
||||
# TODO: besides same code problem, we should also consider other problems lead to retry.
|
||||
if f"{target_task.name}.py" not in batch_edit:
|
||||
continue
|
||||
|
||||
if batch_edit and max(len(i.encode("utf-8")) for i in batch_edit.keys()) > 255:
|
||||
continue
|
||||
|
||||
if batch_edit[f"{target_task.name}.py"] != "__DEL__" and batch_edit[
|
||||
f"{target_task.name}.py"
|
||||
] != workspace.file_dict.get(f"{target_task.name}.py"):
|
||||
break
|
||||
|
||||
# If the task involves model removal, assume it can only process one model at a time.
|
||||
if len(batch_edit) == 1 and batch_edit[f"{target_task.name}.py"] == "__DEL__":
|
||||
break
|
||||
else:
|
||||
raise CoderError("Failed to generate a new model code.")
|
||||
|
||||
return batch_edit
|
||||
|
||||
def assign_code_list_to_evo(self, code_list: list[dict[str, str]], evo):
|
||||
"""
|
||||
Assign the code list to the evolving item.
|
||||
|
||||
The code list is aligned with the evolving item's sub-tasks.
|
||||
If a task is not implemented, put a None in the list.
|
||||
"""
|
||||
for index in range(len(evo.sub_tasks)):
|
||||
if code_list[index] is None:
|
||||
continue
|
||||
if evo.sub_workspace_list[index] is None:
|
||||
# evo.sub_workspace_list[index] = FBWorkspace(target_task=evo.sub_tasks[index])
|
||||
evo.sub_workspace_list[index] = evo.experiment_workspace
|
||||
evo.sub_workspace_list[index].inject_files(**code_list[index])
|
||||
return evo
|
||||
|
||||
|
||||
class ModelCoSTEER(CoSTEER):
|
||||
def __init__(
|
||||
self,
|
||||
scen: Scenario,
|
||||
*args,
|
||||
**kwargs,
|
||||
) -> None:
|
||||
settings = DSCoderCoSTEERSettings()
|
||||
eva = CoSTEERMultiEvaluator(
|
||||
ModelGeneralCaseSpecEvaluator(scen=scen), scen=scen
|
||||
) # Please specify whether you agree running your eva in parallel or not
|
||||
# eva = ModelGeneralCaseSpecEvaluator(scen=scen)
|
||||
es = ModelMultiProcessEvolvingStrategy(scen=scen, settings=settings)
|
||||
|
||||
super().__init__(
|
||||
*args,
|
||||
settings=settings,
|
||||
eva=eva,
|
||||
es=es,
|
||||
evolving_version=2,
|
||||
scen=scen,
|
||||
max_loop=DS_RD_SETTING.coder_max_loop,
|
||||
**kwargs,
|
||||
)
|
||||
@@ -0,0 +1,118 @@
|
||||
"""
|
||||
Beyond previous tests
|
||||
-
|
||||
"""
|
||||
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
from rdagent.app.data_science.conf import DS_RD_SETTING
|
||||
from rdagent.components.coder.CoSTEER.evaluators import (
|
||||
CoSTEEREvaluator,
|
||||
CoSTEERSingleFeedback,
|
||||
)
|
||||
from rdagent.components.coder.data_science.conf import get_ds_env
|
||||
from rdagent.components.coder.data_science.utils import remove_eda_part
|
||||
from rdagent.core.evolving_framework import QueriedKnowledge
|
||||
from rdagent.core.exception import CoderError
|
||||
from rdagent.core.experiment import FBWorkspace, Task
|
||||
from rdagent.oai.llm_utils import APIBackend
|
||||
from rdagent.utils.agent.tpl import T
|
||||
from rdagent.utils.agent.workflow import build_cls_from_json_with_retry
|
||||
|
||||
DIRNAME = Path(__file__).absolute().resolve().parent
|
||||
ModelSingleFeedback = CoSTEERSingleFeedback
|
||||
|
||||
|
||||
# Below are unit tests for testing the specification of the implemented model ------------------
|
||||
class ModelGeneralCaseSpecEvaluator(CoSTEEREvaluator):
|
||||
"""
|
||||
Motivation case:
|
||||
- Simplest case, we already split the data into train_data, valid_data, and test_data. We require the model to learn (optionally validate on valid data), and infer on test data.
|
||||
|
||||
Test workflow:
|
||||
- Build train, valid, and test data to run it, and test the output (e.g., shape, etc.)
|
||||
"""
|
||||
|
||||
def evaluate(
|
||||
self,
|
||||
target_task: Task,
|
||||
implementation: FBWorkspace,
|
||||
gt_implementation: FBWorkspace,
|
||||
queried_knowledge: QueriedKnowledge = None,
|
||||
**kwargs,
|
||||
) -> ModelSingleFeedback:
|
||||
target_task_information = target_task.get_task_information()
|
||||
if (
|
||||
queried_knowledge is not None
|
||||
and target_task_information in queried_knowledge.success_task_to_knowledge_dict
|
||||
):
|
||||
return queried_knowledge.success_task_to_knowledge_dict[target_task_information].feedback
|
||||
elif queried_knowledge is not None and target_task_information in queried_knowledge.failed_task_info_set:
|
||||
return ModelSingleFeedback(
|
||||
execution="This task has failed too many times, skip implementation.",
|
||||
return_checking="This task has failed too many times, skip implementation.",
|
||||
code="This task has failed too many times, skip implementation.",
|
||||
final_decision=False,
|
||||
)
|
||||
|
||||
env = get_ds_env(extra_volumes={self.scen.debug_path: T("scenarios.data_science.share:scen.input_path").r()})
|
||||
|
||||
if_model_removed = False
|
||||
|
||||
if f"{target_task.name}.py" in implementation.file_dict:
|
||||
fname = "test/model_test.py"
|
||||
test_code = (
|
||||
(DIRNAME / "eval_tests" / "model_test.txt").read_text().replace("model01", target_task.name)
|
||||
) # only check the model changed this time
|
||||
implementation.inject_files(**{fname: test_code})
|
||||
stdout, ret_code = implementation.execute_ret_code(env=env, entry=f"python {fname}")
|
||||
|
||||
if stdout is None:
|
||||
raise CoderError(
|
||||
"The execution output contains too many progress bars and results in the LLM's token size exceeding the limit."
|
||||
)
|
||||
else:
|
||||
ret_code = 0
|
||||
if_model_removed = True
|
||||
stdout = f"Model {target_task.name} removal succeeded."
|
||||
|
||||
if "main.py" in implementation.file_dict and ret_code == 0:
|
||||
workflow_stdout = implementation.execute(env=env, entry="python main.py")
|
||||
workflow_stdout = remove_eda_part(workflow_stdout)
|
||||
else:
|
||||
workflow_stdout = None
|
||||
|
||||
if if_model_removed:
|
||||
system_prompt = T(".prompts:model_eval_rm.system").r(
|
||||
task_desc=target_task.get_task_information(),
|
||||
workflow_stdout=workflow_stdout,
|
||||
workflow_code=implementation.all_codes,
|
||||
)
|
||||
user_prompt = T(".prompts:model_eval_rm.user").r(
|
||||
stdout=stdout,
|
||||
workflow_stdout=workflow_stdout,
|
||||
)
|
||||
else:
|
||||
system_prompt = T(".prompts:model_eval.system").r(
|
||||
task_desc=target_task.get_task_information(),
|
||||
test_code=test_code,
|
||||
code=implementation.file_dict[f"{target_task.name}.py"],
|
||||
workflow_stdout=workflow_stdout,
|
||||
workflow_code=implementation.all_codes,
|
||||
)
|
||||
user_prompt = T(".prompts:model_eval.user").r(
|
||||
stdout=stdout,
|
||||
workflow_stdout=workflow_stdout,
|
||||
)
|
||||
|
||||
fb = build_cls_from_json_with_retry(
|
||||
ModelSingleFeedback,
|
||||
system_prompt=system_prompt,
|
||||
user_prompt=user_prompt,
|
||||
init_kwargs_update_func=ModelSingleFeedback.val_and_update_init_dict,
|
||||
)
|
||||
fb.final_decision = fb.final_decision and ret_code == 0
|
||||
|
||||
return fb
|
||||
@@ -0,0 +1,105 @@
|
||||
"""
|
||||
Tests for `model_workflow` in model01.py
|
||||
"""
|
||||
import sys
|
||||
import time
|
||||
|
||||
from feature import feat_eng
|
||||
from load_data import load_data
|
||||
from model01 import model_workflow
|
||||
from sklearn.model_selection import train_test_split
|
||||
|
||||
|
||||
def log_execution_results(start_time, val_pred, test_pred, hypers, execution_label):
|
||||
"""Log the results of a single model execution."""
|
||||
feedback_str = f"{execution_label} end.\n"
|
||||
feedback_str += f"Validation predictions shape: {val_pred.shape if val_pred is not None else 'None'}\n"
|
||||
feedback_str += f"Test predictions shape: {test_pred.shape if test_pred is not None else 'None'}\n"
|
||||
feedback_str += f"Hyperparameters: {hypers if hypers is not None else 'None'}\n"
|
||||
feedback_str += f"Execution time: {time.time() - start_time:.2f} seconds.\n"
|
||||
print(feedback_str)
|
||||
|
||||
|
||||
import reprlib
|
||||
aRepr = reprlib.Repr()
|
||||
aRepr.maxother=300
|
||||
|
||||
# Load and preprocess data
|
||||
X, y, test_X, test_ids = load_data()
|
||||
X, y, test_X = feat_eng(X, y, test_X)
|
||||
|
||||
print(f"X.shape: {X.shape}" if hasattr(X, 'shape') else f"X length: {len(X)}")
|
||||
print(f"y.shape: {y.shape}" if hasattr(y, 'shape') else f"y length: {len(y)}")
|
||||
print(f"test_X.shape: {test_X.shape}" if hasattr(test_X, 'shape') else f"test_X length: {len(test_X)}")
|
||||
print(f"test_ids length: {len(test_ids)}")
|
||||
|
||||
train_X, val_X, train_y, val_y = train_test_split(X, y, test_size=0.8, random_state=42)
|
||||
|
||||
|
||||
import sys
|
||||
import reprlib
|
||||
from joblib.memory import MemorizedFunc
|
||||
|
||||
|
||||
def get_original_code(func):
|
||||
if isinstance(func, MemorizedFunc):
|
||||
return func.func.__code__
|
||||
return func.__code__
|
||||
|
||||
print("train_X:", aRepr.repr(train_X))
|
||||
print("train_y:", aRepr.repr(train_y))
|
||||
print("val_X:", aRepr.repr(val_X))
|
||||
print("val_y:", aRepr.repr(val_y))
|
||||
|
||||
print(f"train_X.shape: {train_X.shape}" if hasattr(train_X, 'shape') else f"train_X length: {len(train_X)}")
|
||||
print(f"train_y.shape: {train_y.shape}" if hasattr(train_y, 'shape') else f"train_y length: {len(train_y)}")
|
||||
print(f"val_X.shape: {val_X.shape}" if hasattr(val_X, 'shape') else f"val_X length: {len(val_X)}")
|
||||
print(f"val_y.shape: {val_y.shape}" if hasattr(val_y, 'shape') else f"val_y length: {len(val_y)}")
|
||||
|
||||
|
||||
|
||||
def debug_info_print(func):
|
||||
def wrapper(*args, **kwargs):
|
||||
original_code = get_original_code(func)
|
||||
def local_trace(frame, event, arg):
|
||||
if event == "return" and frame.f_code == original_code:
|
||||
print("\n" + "="*20 + "Running model training code, local variable values:" + "="*20)
|
||||
for k, v in frame.f_locals.items():
|
||||
printed = aRepr.repr(v)
|
||||
print(f"{k}:\n {printed}")
|
||||
print("="*20 + "Local variable values end" + "="*20)
|
||||
return local_trace
|
||||
|
||||
sys.settrace(local_trace)
|
||||
try:
|
||||
return func(*args, **kwargs)
|
||||
finally:
|
||||
sys.settrace(None)
|
||||
return wrapper
|
||||
|
||||
# First execution
|
||||
print("The first execution begins.\n")
|
||||
start_time = time.time()
|
||||
val_pred, test_pred, hypers = debug_info_print(model_workflow)(
|
||||
X=train_X,
|
||||
y=train_y,
|
||||
val_X=val_X,
|
||||
val_y=val_y,
|
||||
test_X=None,
|
||||
)
|
||||
log_execution_results(start_time, val_pred, test_pred, hypers, "The first execution")
|
||||
|
||||
# Second execution
|
||||
print("The second execution begins.\n")
|
||||
start_time = time.time()
|
||||
val_pred, test_pred, final_hypers = debug_info_print(model_workflow)(
|
||||
X=train_X,
|
||||
y=train_y,
|
||||
val_X=None,
|
||||
val_y=None,
|
||||
test_X=test_X,
|
||||
hyper_params=hypers,
|
||||
)
|
||||
log_execution_results(start_time, val_pred, test_pred, final_hypers, "The second execution")
|
||||
|
||||
print("Model code test end.")
|
||||
@@ -0,0 +1,21 @@
|
||||
from typing import Dict, Optional
|
||||
|
||||
from rdagent.components.coder.CoSTEER.task import CoSTEERTask
|
||||
|
||||
|
||||
# Because we use isinstance to distinguish between different types of tasks, we need to use sub classes to represent different types of tasks
|
||||
class ModelTask(CoSTEERTask):
|
||||
def __init__(
|
||||
self,
|
||||
name: str,
|
||||
description: str,
|
||||
*args,
|
||||
**kwargs,
|
||||
) -> None:
|
||||
super().__init__(name=name, description=description, *args, **kwargs)
|
||||
|
||||
def get_task_information(self):
|
||||
task_desc = f"""name: {self.name}
|
||||
description: {self.description}
|
||||
"""
|
||||
return task_desc
|
||||
@@ -0,0 +1,186 @@
|
||||
model_coder:
|
||||
system: |-
|
||||
You are a world-class data scientist and machine learning engineer with deep expertise in statistics, mathematics, and computer science.
|
||||
Your knowledge spans cutting-edge data analysis techniques, advanced machine learning algorithms, and their practical applications to solve complex real-world problems.
|
||||
|
||||
## Task Description
|
||||
{{ task_desc }}
|
||||
|
||||
## Competition Information for This Task
|
||||
{{ competition_info }}
|
||||
|
||||
{% if queried_similar_successful_knowledge|length != 0 or queried_former_failed_knowledge|length != 0 %}
|
||||
## Relevant Information for This Task
|
||||
{% endif %}
|
||||
|
||||
{% if queried_similar_successful_knowledge|length != 0 %}
|
||||
--------- Successful Implementations for Similar Models ---------
|
||||
====={% for similar_successful_knowledge in queried_similar_successful_knowledge %} Model {{ loop.index }}:=====
|
||||
{{ similar_successful_knowledge.target_task.get_task_information() }}
|
||||
=====Code:=====
|
||||
{{ similar_successful_knowledge.implementation.file_dict[similar_successful_knowledge.target_task.name ~ '.py'] }}
|
||||
{% endfor %}
|
||||
{% endif %}
|
||||
|
||||
{% if queried_former_failed_knowledge|length != 0 %}
|
||||
--------- Previous Failed Attempts ---------
|
||||
{% for former_failed_knowledge in queried_former_failed_knowledge %} Attempt {{ loop.index }}:
|
||||
=====Code:=====
|
||||
{{ former_failed_knowledge.implementation.file_dict[former_failed_knowledge.target_task.name ~ '.py'] }}
|
||||
=====Feedback:=====
|
||||
{{ former_failed_knowledge.feedback }}
|
||||
{% endfor %}
|
||||
{% endif %}
|
||||
|
||||
## Guidelines
|
||||
1. The function's input is from the output of a feature engineering function whose input is the output of a data loading function. The data loader function and feature engineering function code is as follows:
|
||||
--------- Data Loader Code ---------
|
||||
{{ data_loader_code }}
|
||||
--------- Feature Engineering Code ---------
|
||||
{{ feature_code }}
|
||||
2. You should avoid using logging module to output information in your generated code, and instead use the print() function.
|
||||
3. If the model can both be implemented by PyTorch and Tensorflow, please use pytorch for broader compatibility.
|
||||
4. You should use the following cache decorator to cache the results of the function:
|
||||
```python
|
||||
from joblib import Memory
|
||||
memory = Memory(location='{% include "scenarios.data_science.share:scen.cache_path" %}', verbose=0)
|
||||
@memory.cache``
|
||||
{% include "scenarios.data_science.share:guidelines.coding" %}
|
||||
|
||||
## Output Format
|
||||
{% if out_spec %}
|
||||
{{ out_spec }}
|
||||
The file name should be the model name described in the model task in the format "{task_name}.py". You should always follow this name format.
|
||||
{% else %}
|
||||
Please response the code in the following json format. Here is an example structure for the JSON output:
|
||||
{
|
||||
"code": "The Python code as a string."
|
||||
}
|
||||
{% endif %}
|
||||
|
||||
user_general: |-
|
||||
--------- Code Specification ---------
|
||||
{{ code_spec }}
|
||||
|
||||
--------- Former model code ---------
|
||||
{% if latest_model_code|length == 0 %}
|
||||
So far the workspace is empty. No model code has been implemented yet.
|
||||
{% else %}
|
||||
{{ latest_model_code }}
|
||||
{% if latest_code_feedback is not none %}
|
||||
--------- Feedback to former code ---------
|
||||
{{ latest_code_feedback }}
|
||||
{% endif %}
|
||||
{% endif %}
|
||||
|
||||
model_eval:
|
||||
system: |-
|
||||
You are a data scientist responsible for evaluating model building code generation.
|
||||
|
||||
## Task Description
|
||||
{{ task_desc }}
|
||||
|
||||
## Model Building Code
|
||||
```python
|
||||
{{ code }}
|
||||
```
|
||||
|
||||
## Testing Process
|
||||
The model building code is tested using the following script:
|
||||
```python
|
||||
{{ test_code }}
|
||||
```
|
||||
|
||||
### Execution Phases
|
||||
The model is tested in two phases:
|
||||
|
||||
1. Initial Training Phase:
|
||||
- The model receives **train and valid inputs** with **empty hyperparameters**.
|
||||
- The focus is on verifying whether the model successfully trains and produces **valid outputs and hyperparameter outputs**.
|
||||
|
||||
2. Retraining Phase:
|
||||
- The model receives **train and test inputs** (without valid inputs).
|
||||
- The hyperparameters generated from the first phase are passed back for **retraining**.
|
||||
|
||||
|
||||
### Key Requirements for Approval
|
||||
A model can only be approved if it meets all of the following conditions:
|
||||
1. Hyperparameter Handling
|
||||
- If hyperparameters are returned, they must include an early stop round.
|
||||
- The hyperparameters must be correctly utilized in the model for retraining.
|
||||
- If the early stop round is provided, it must be used in the model implementation.
|
||||
2. The model output shape must strictly match the specifications in `spec.md`.
|
||||
|
||||
{% if workflow_stdout is not none %}
|
||||
### Whole Workflow Consideration
|
||||
The model building code is part of the whole workflow. The user has executed the entire pipeline and provided additional stdout.
|
||||
|
||||
**Workflow Code:**
|
||||
```python
|
||||
{{ workflow_code }}
|
||||
```
|
||||
|
||||
You should evaluate both the model building test results and the overall workflow results. **Approve the code only if both tests pass.**
|
||||
{% endif %}
|
||||
|
||||
## Evaluation Criteria
|
||||
You will be given the standard output (`stdout`) from the model building test and, if applicable, the workflow test.
|
||||
[Note] If no stdout for model buidling test is provided, the model failed due to a timeout or out-of-memory error. You should analyze potential optimizations.
|
||||
|
||||
Please respond with your feedback in the following JSON format and order
|
||||
```json
|
||||
{
|
||||
"execution": "Describe how well the model building executed, including any errors or issues encountered. Append all error messages and full traceback details without summarizing or omitting any information.",
|
||||
"return_checking": "Check the generated value, including whether the value is generated and comparing the shape of the model output with the requirement in spec.md. You also need to check whether the hyperparameters used for retraining are correctly returned during the test execution of the model.",
|
||||
"code": "Assess code quality, readability, and adherence to specifications. Consider efficiency, including whether the code utilizes multi-threading or GPU acceleration for optimization.",
|
||||
"final_decision": <true/false>
|
||||
}
|
||||
```
|
||||
|
||||
user: |-
|
||||
--------- Model building test stdout ---------
|
||||
{{ stdout }}
|
||||
{% if workflow_stdout is not none %}
|
||||
--------- Whole workflow test stdout ---------
|
||||
{{ workflow_stdout }}
|
||||
{% endif %}
|
||||
|
||||
model_eval_rm:
|
||||
system: |-
|
||||
You are a data scientist responsible for evaluating model removal process.
|
||||
|
||||
## Task Description
|
||||
{{ task_desc }}
|
||||
|
||||
{% if workflow_stdout is not none %}
|
||||
## Whole Workflow Consideration
|
||||
The model building code is part of the whole workflow. The user has executed the entire pipeline and provided additional stdout.
|
||||
|
||||
**Workflow Code:**
|
||||
```python
|
||||
{{ workflow_code }}
|
||||
```
|
||||
|
||||
You should evaluate both the model removal test results and the overall workflow results. **Approve the code only if both tests pass.**
|
||||
{% endif %}
|
||||
|
||||
## Evaluation Criteria
|
||||
You will be given the standard output (`stdout`) from the model removal test and, if applicable, the workflow test.
|
||||
|
||||
Please respond with your feedback in the following JSON format and order
|
||||
```json
|
||||
{
|
||||
"execution": "Describe how well the model removal executed, including any errors or issues encountered. Append all error messages and full traceback details without summarizing or omitting any information.",
|
||||
"return_checking": "Check the generated value, including whether the value is generated and comparing the shape of the model output with the requirement in spec.md.",
|
||||
"code": "Assess code quality, readability, and adherence to specifications.",
|
||||
"final_decision": <true/false>
|
||||
}
|
||||
```
|
||||
|
||||
user: |-
|
||||
--------- Model removal test stdout ---------
|
||||
{{ stdout }}
|
||||
{% if workflow_stdout is not none %}
|
||||
--------- Whole workflow test stdout ---------
|
||||
{{ workflow_stdout }}
|
||||
{% endif %}
|
||||
@@ -0,0 +1,67 @@
|
||||
"""
|
||||
Generate dataset to test the model workflow output
|
||||
"""
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from rdagent.components.coder.CoSTEER.config import CoSTEER_SETTINGS
|
||||
from rdagent.components.coder.data_science.model import ModelCoSTEER
|
||||
from rdagent.components.coder.data_science.model.eval import (
|
||||
ModelGeneralCaseSpecEvaluator,
|
||||
)
|
||||
from rdagent.components.coder.data_science.model.exp import ModelTask
|
||||
from rdagent.core.experiment import FBWorkspace
|
||||
from rdagent.scenarios.data_science.experiment.experiment import DSExperiment
|
||||
from rdagent.scenarios.data_science.scen import KaggleScen
|
||||
|
||||
|
||||
# Take tasks, spec.md and feat as input, generate a feedback as output
|
||||
def develop_one_competition(competition: str):
|
||||
scen = KaggleScen(competition=competition)
|
||||
model_coder = ModelCoSTEER(scen)
|
||||
|
||||
# Create the task
|
||||
mt = ModelTask(
|
||||
name="ModelTask",
|
||||
description="A CNN Model",
|
||||
model_type="CNN",
|
||||
architecture="\hat{y}_u = CNN(X_u)",
|
||||
# variables="variables: {'\\hat{y}_u': 'The predicted output for node u', 'X_u': 'The input features for node u'}",
|
||||
hyperparameters="...",
|
||||
base_code="",
|
||||
)
|
||||
|
||||
tpl_ex_path = Path(__file__).resolve() / Path("rdagent/scenarios/kaggle/tpl_ex").resolve() / competition
|
||||
injected_file_names = ["spec/model.md", "load_data.py", "feature.py", "model01.py"]
|
||||
|
||||
modelexp = FBWorkspace()
|
||||
for file_name in injected_file_names:
|
||||
file_path = tpl_ex_path / file_name
|
||||
modelexp.inject_files(**{file_name: file_path.read_text()})
|
||||
|
||||
mt.base_code += modelexp.file_dict["model01.py"]
|
||||
exp = DSExperiment(
|
||||
sub_tasks=[mt],
|
||||
)
|
||||
|
||||
# Test the evaluator:
|
||||
"""eva = ModelGeneralCaseSpecEvaluator(scen=scen)
|
||||
exp.feedback = eva.evaluate(target_task=mt, queried_knowledge=None, implementation=modelexp, gt_implementation=None)
|
||||
print(exp.feedback)"""
|
||||
|
||||
# Test the evolving strategy:
|
||||
"""es = ModelMultiProcessEvolvingStrategy(scen=scen, settings=CoSTEER_SETTINGS)
|
||||
new_code = es.implement_one_task(target_task=mt, queried_knowledge=None, workspace=modelexp)
|
||||
print(new_code)"""
|
||||
|
||||
# Run the experiment
|
||||
for file_name in injected_file_names:
|
||||
file_path = tpl_ex_path / file_name
|
||||
exp.experiment_workspace.inject_files(**{file_name: file_path.read_text()})
|
||||
|
||||
exp = model_coder.develop(exp)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
develop_one_competition("aerial-cactus-identification")
|
||||
# dotenv run -- python rdagent/components/coder/data_science/model/test.py
|
||||
@@ -0,0 +1,178 @@
|
||||
"""
|
||||
|
||||
Loop should not large change exclude
|
||||
- Action Choice[current data loader & spec]
|
||||
- other should share
|
||||
- Propose[choice] => Task[Choice] => CoSTEER =>
|
||||
-
|
||||
|
||||
Extra feature:
|
||||
- cache
|
||||
|
||||
|
||||
File structure
|
||||
- ___init__.py: the entrance/agent of coder
|
||||
- evaluator.py
|
||||
- conf.py
|
||||
- exp.py: everything under the experiment, e.g.
|
||||
- Task
|
||||
- Experiment
|
||||
- Workspace
|
||||
- test.py
|
||||
- Each coder could be tested.
|
||||
"""
|
||||
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Dict
|
||||
|
||||
from rdagent.app.data_science.conf import DS_RD_SETTING
|
||||
from rdagent.components.coder.CoSTEER import CoSTEER
|
||||
from rdagent.components.coder.CoSTEER.evaluators import (
|
||||
CoSTEERMultiEvaluator,
|
||||
CoSTEERSingleFeedback,
|
||||
)
|
||||
from rdagent.components.coder.CoSTEER.evolving_strategy import (
|
||||
MultiProcessEvolvingStrategy,
|
||||
)
|
||||
from rdagent.components.coder.CoSTEER.knowledge_management import (
|
||||
CoSTEERQueriedKnowledge,
|
||||
)
|
||||
from rdagent.components.coder.data_science.conf import (
|
||||
DSCoderCoSTEERSettings,
|
||||
get_ds_env,
|
||||
)
|
||||
from rdagent.components.coder.data_science.pipeline.eval import PipelineCoSTEEREvaluator
|
||||
from rdagent.components.coder.data_science.raw_data_loader.eval import (
|
||||
DataLoaderCoSTEEREvaluator,
|
||||
)
|
||||
from rdagent.components.coder.data_science.raw_data_loader.exp import DataLoaderTask
|
||||
from rdagent.components.coder.data_science.share.eval import ModelDumpEvaluator
|
||||
from rdagent.core.exception import CoderError
|
||||
from rdagent.core.experiment import FBWorkspace
|
||||
from rdagent.core.scenario import Scenario
|
||||
from rdagent.oai.llm_utils import APIBackend
|
||||
from rdagent.utils.agent.ret import PythonAgentOut
|
||||
from rdagent.utils.agent.tpl import T
|
||||
|
||||
DIRNAME = Path(__file__).absolute().resolve().parent
|
||||
|
||||
|
||||
class PipelineMultiProcessEvolvingStrategy(MultiProcessEvolvingStrategy):
|
||||
def implement_one_task(
|
||||
self,
|
||||
target_task: DataLoaderTask,
|
||||
queried_knowledge: CoSTEERQueriedKnowledge | None = None,
|
||||
workspace: FBWorkspace | None = None,
|
||||
prev_task_feedback: CoSTEERSingleFeedback | None = None,
|
||||
) -> dict[str, str]:
|
||||
competition_info = self.scen.get_scenario_all_desc(eda_output=workspace.file_dict.get("EDA.md", None))
|
||||
runtime_environment = self.scen.get_runtime_environment()
|
||||
data_folder_info = self.scen.processed_data_folder_description
|
||||
pipeline_task_info = target_task.get_task_information()
|
||||
|
||||
queried_similar_successful_knowledge = (
|
||||
queried_knowledge.task_to_similar_task_successful_knowledge[pipeline_task_info]
|
||||
if queried_knowledge is not None
|
||||
else []
|
||||
)
|
||||
queried_former_failed_knowledge = (
|
||||
queried_knowledge.task_to_former_failed_traces[pipeline_task_info] if queried_knowledge is not None else []
|
||||
)
|
||||
queried_former_failed_knowledge = (
|
||||
[
|
||||
knowledge
|
||||
for knowledge in queried_former_failed_knowledge[0]
|
||||
if knowledge.implementation.file_dict.get("main.py") != workspace.file_dict.get("main.py")
|
||||
],
|
||||
queried_former_failed_knowledge[1],
|
||||
)
|
||||
|
||||
system_prompt = T(".prompts:pipeline_coder.system").r(
|
||||
task_desc=pipeline_task_info,
|
||||
queried_similar_successful_knowledge=queried_similar_successful_knowledge,
|
||||
queried_former_failed_knowledge=queried_former_failed_knowledge[0],
|
||||
out_spec=PythonAgentOut.get_spec(),
|
||||
runtime_environment=runtime_environment,
|
||||
spec=T("scenarios.data_science.share:component_spec.Pipeline").r(),
|
||||
enable_model_dump=DS_RD_SETTING.enable_model_dump,
|
||||
)
|
||||
if DS_RD_SETTING.proposal_version == "v3":
|
||||
# FIXME: A temporary patch for BUILD
|
||||
user_prompt = T(".prompts:pipeline_coder.user_v3").r(
|
||||
competition_info=competition_info,
|
||||
folder_spec=data_folder_info,
|
||||
latest_code=workspace.file_dict.get("main.py"),
|
||||
latest_code_feedback=prev_task_feedback,
|
||||
)
|
||||
else:
|
||||
user_prompt = T(".prompts:pipeline_coder.user").r(
|
||||
competition_info=competition_info,
|
||||
folder_spec=data_folder_info,
|
||||
latest_code=workspace.file_dict.get("main.py"),
|
||||
latest_code_feedback=prev_task_feedback,
|
||||
)
|
||||
|
||||
for _ in range(5):
|
||||
pipeline_code = PythonAgentOut.extract_output(
|
||||
APIBackend().build_messages_and_create_chat_completion(
|
||||
user_prompt=user_prompt,
|
||||
system_prompt=system_prompt,
|
||||
)
|
||||
)
|
||||
if pipeline_code != workspace.file_dict.get("main.py"):
|
||||
break
|
||||
else:
|
||||
user_prompt = user_prompt + "\nPlease avoid generating same code to former code!"
|
||||
else:
|
||||
raise CoderError("Failed to generate a new pipeline code.")
|
||||
|
||||
return {
|
||||
"main.py": pipeline_code,
|
||||
}
|
||||
|
||||
def assign_code_list_to_evo(self, code_list: list[dict[str, str]], evo):
|
||||
"""
|
||||
Assign the code list to the evolving item.
|
||||
|
||||
The code list is aligned with the evolving item's sub-tasks.
|
||||
If a task is not implemented, put a None in the list.
|
||||
"""
|
||||
for index in range(len(evo.sub_tasks)):
|
||||
if code_list[index] is None:
|
||||
continue
|
||||
if evo.sub_workspace_list[index] is None:
|
||||
# evo.sub_workspace_list[index] = FBWorkspace(target_task=evo.sub_tasks[index])
|
||||
evo.sub_workspace_list[index] = evo.experiment_workspace
|
||||
evo.sub_workspace_list[index].inject_files(**code_list[index])
|
||||
return evo
|
||||
|
||||
|
||||
class PipelineCoSTEER(CoSTEER):
|
||||
def __init__(
|
||||
self,
|
||||
scen: Scenario,
|
||||
*args,
|
||||
**kwargs,
|
||||
) -> None:
|
||||
settings = DSCoderCoSTEERSettings()
|
||||
eval_l = [PipelineCoSTEEREvaluator(scen=scen)]
|
||||
if DS_RD_SETTING.enable_model_dump:
|
||||
eval_l.append(ModelDumpEvaluator(scen=scen, data_type="sample"))
|
||||
|
||||
eva = CoSTEERMultiEvaluator(
|
||||
single_evaluator=eval_l, scen=scen
|
||||
) # Please specify whether you agree running your eva in parallel or not
|
||||
es = PipelineMultiProcessEvolvingStrategy(scen=scen, settings=settings)
|
||||
|
||||
super().__init__(
|
||||
*args,
|
||||
settings=settings,
|
||||
eva=eva,
|
||||
es=es,
|
||||
evolving_version=2,
|
||||
scen=scen,
|
||||
max_loop=DS_RD_SETTING.coder_max_loop,
|
||||
**kwargs,
|
||||
)
|
||||
@@ -0,0 +1,159 @@
|
||||
# tess successfully running.
|
||||
# (GPT) if it aligns with the spec & rationality of the spec.
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from rdagent.app.data_science.conf import DS_RD_SETTING
|
||||
from rdagent.components.coder.CoSTEER import CoSTEERMultiFeedback
|
||||
from rdagent.components.coder.CoSTEER.evaluators import (
|
||||
CoSTEEREvaluator,
|
||||
CoSTEERSingleFeedback,
|
||||
)
|
||||
from rdagent.components.coder.CoSTEER.knowledge_management import (
|
||||
CoSTEERQueriedKnowledgeV2,
|
||||
)
|
||||
from rdagent.components.coder.data_science.conf import get_clear_ws_cmd, get_ds_env
|
||||
from rdagent.components.coder.data_science.utils import remove_eda_part
|
||||
from rdagent.core.experiment import FBWorkspace, Task
|
||||
from rdagent.scenarios.data_science.test_eval import get_test_eval
|
||||
from rdagent.utils.agent.tpl import T
|
||||
from rdagent.utils.agent.workflow import build_cls_from_json_with_retry
|
||||
|
||||
DIRNAME = Path(__file__).absolute().resolve().parent
|
||||
|
||||
PipelineSingleFeedback = CoSTEERSingleFeedback
|
||||
PipelineMultiFeedback = CoSTEERMultiFeedback
|
||||
|
||||
|
||||
class PipelineCoSTEEREvaluator(CoSTEEREvaluator):
|
||||
|
||||
def evaluate(
|
||||
self,
|
||||
target_task: Task,
|
||||
implementation: FBWorkspace,
|
||||
gt_implementation: FBWorkspace,
|
||||
queried_knowledge: CoSTEERQueriedKnowledgeV2 = None,
|
||||
**kwargs,
|
||||
) -> PipelineSingleFeedback:
|
||||
|
||||
target_task_information = target_task.get_task_information()
|
||||
if (
|
||||
queried_knowledge is not None
|
||||
and target_task_information in queried_knowledge.success_task_to_knowledge_dict
|
||||
):
|
||||
return queried_knowledge.success_task_to_knowledge_dict[target_task_information].feedback
|
||||
elif queried_knowledge is not None and target_task_information in queried_knowledge.failed_task_info_set:
|
||||
return PipelineSingleFeedback(
|
||||
execution="This task has failed too many times, skip implementation.",
|
||||
return_checking="This task has failed too many times, skip implementation.",
|
||||
code="This task has failed too many times, skip implementation.",
|
||||
final_decision=False,
|
||||
)
|
||||
|
||||
env = get_ds_env(extra_volumes={self.scen.debug_path: T("scenarios.data_science.share:scen.input_path").r()})
|
||||
|
||||
# Clean the scores.csv & submission.csv.
|
||||
implementation.execute(env=env, entry=get_clear_ws_cmd())
|
||||
stdout, execute_ret_code = implementation.execute_ret_code(env=env, entry=f"python -m coverage run main.py")
|
||||
stdout = remove_eda_part(stdout)
|
||||
stdout += f"The code executed {'successfully' if execute_ret_code == 0 else 'failed'}."
|
||||
|
||||
score_fp = implementation.workspace_path / "scores.csv"
|
||||
score_ret_code = 0
|
||||
score_check_text = ""
|
||||
if not score_fp.exists():
|
||||
score_check_text = "[Error] Metrics file (scores.csv) is not generated!"
|
||||
score_ret_code = 1
|
||||
else:
|
||||
try:
|
||||
score_df = pd.read_csv(score_fp, index_col=0)
|
||||
model_set_in_scores = set(score_df.index)
|
||||
|
||||
# Check model names (index)
|
||||
if not score_df.index.is_unique:
|
||||
score_check_text += "\n[Error] The score dataframe contains duplicate model names."
|
||||
score_ret_code = 1
|
||||
if "ensemble" not in model_set_in_scores:
|
||||
score_check_text += "\n[Error] The score dataframe doesn't contain the ensemble model."
|
||||
score_ret_code = 1
|
||||
if score_ret_code != 0:
|
||||
score_check_text += f"The score_df is:\n{score_df}"
|
||||
|
||||
# Check metric name (columns)
|
||||
if score_df.columns.tolist() != [self.scen.metric_name]:
|
||||
score_check_text += f"\n[Error] The scores dataframe does not contain the correct column names.\nCorrect columns is: ['{self.scen.metric_name}']\nBut got: {score_df.columns.tolist()}"
|
||||
score_ret_code = 1
|
||||
|
||||
# Check if scores contain NaN (values)
|
||||
if score_df.isnull().values.any():
|
||||
nan_locations = score_df[score_df.isnull().any(axis=1)]
|
||||
score_check_text += f"\n[Error] The scores dataframe contains NaN values at the following locations:\n{nan_locations}"
|
||||
score_ret_code = 1
|
||||
|
||||
except Exception as e:
|
||||
score_check_text += f"\n[Error] in checking the scores.csv file: {e}\nscores.csv's content:\n-----\n{score_fp.read_text()}\n-----"
|
||||
score_ret_code = 1
|
||||
|
||||
test_eval = get_test_eval()
|
||||
if not test_eval.is_sub_enabled(self.scen.competition):
|
||||
submission_ret_code = 0
|
||||
else:
|
||||
# Check submission file
|
||||
base_check_code = T(".eval_tests.submission_format_test", ftype="txt").r()
|
||||
implementation.inject_files(**{"test/submission_format_test.py": base_check_code})
|
||||
# stdout += "----Submission Check 1-----\n"
|
||||
submission_check_out, submission_ret_code = implementation.execute_ret_code(
|
||||
env=env, entry="python test/submission_format_test.py"
|
||||
)
|
||||
if DS_RD_SETTING.rule_base_eval:
|
||||
if execute_ret_code == 0 and score_ret_code == 0 and submission_ret_code == 0:
|
||||
return PipelineSingleFeedback(
|
||||
execution=stdout,
|
||||
return_checking=score_check_text + "\n" + submission_check_out,
|
||||
code="Code evaluation is not available.",
|
||||
final_decision=True,
|
||||
)
|
||||
else:
|
||||
return PipelineSingleFeedback(
|
||||
execution=stdout,
|
||||
return_checking=score_check_text + "\n" + submission_check_out,
|
||||
code="Code evaluation is not available.",
|
||||
final_decision=False,
|
||||
)
|
||||
stdout += "\n" + submission_check_out
|
||||
|
||||
eda_output = implementation.file_dict.get("EDA.md", None)
|
||||
|
||||
eda_output = implementation.file_dict.get("EDA.md", None)
|
||||
|
||||
if not isinstance(implementation, FBWorkspace):
|
||||
eda_output = None
|
||||
else:
|
||||
eda_output = implementation.file_dict.get("EDA.md", None)
|
||||
|
||||
system_prompt = T(".prompts:pipeline_eval.system").r(
|
||||
scenario=self.scen.get_scenario_all_desc(eda_output=eda_output),
|
||||
task_desc=target_task.get_task_information(),
|
||||
is_sub_enabled=test_eval.is_sub_enabled(self.scen.competition),
|
||||
spec=T("scenarios.data_science.share:component_spec.Pipeline").r(),
|
||||
)
|
||||
user_prompt = T(".prompts:pipeline_eval.user").r(
|
||||
stdout=stdout.strip(),
|
||||
code=implementation.file_dict["main.py"],
|
||||
)
|
||||
wfb = build_cls_from_json_with_retry(
|
||||
PipelineSingleFeedback,
|
||||
system_prompt=system_prompt,
|
||||
user_prompt=user_prompt,
|
||||
init_kwargs_update_func=PipelineSingleFeedback.val_and_update_init_dict,
|
||||
)
|
||||
if score_ret_code != 0:
|
||||
wfb.final_decision = False
|
||||
wfb.return_checking += "\n" + score_check_text
|
||||
if submission_ret_code != 0:
|
||||
wfb.final_decision = False
|
||||
wfb.return_checking += "\nSubmission file check failed."
|
||||
return wfb
|
||||
@@ -0,0 +1,86 @@
|
||||
import hashlib
|
||||
from pathlib import Path
|
||||
|
||||
import pandas as pd
|
||||
|
||||
|
||||
def calculate_md5(file_path):
|
||||
with open(file_path, "rb") as f:
|
||||
file_hash = hashlib.md5(f.read()).hexdigest()
|
||||
return file_hash
|
||||
|
||||
|
||||
file_md5 = calculate_md5("scores.csv")
|
||||
|
||||
"""
|
||||
find . | grep -i sample | grep -i submission | grep -v sample_submission.csv | grep -v zip_files | grep -v 'sample/'
|
||||
./denoising-dirty-documents/sampleSubmission.csv
|
||||
./the-icml-2013-whale-challenge-right-whale-redux/sampleSubmission.csv
|
||||
./text-normalization-challenge-russian-language/ru_sample_submission_2.csv.zip
|
||||
./text-normalization-challenge-russian-language/ru_sample_submission_2.csv
|
||||
./random-acts-of-pizza/sampleSubmission.csv
|
||||
./text-normalization-challenge-english-language/en_sample_submission_2.csv.zip
|
||||
./text-normalization-challenge-english-language/en_sample_submission_2.csv
|
||||
./detecting-insults-in-social-commentary/sample_submission_null.csv
|
||||
"""
|
||||
|
||||
# Find sample submission file dynamically
|
||||
input_dir = Path("{% include "scenarios.data_science.share:scen.input_path" %}")
|
||||
# Look for common variations of sample submission filenames
|
||||
sample_submission_files = list(input_dir.glob("*sample_submission*.csv")) + list(
|
||||
input_dir.glob("*sampleSubmission*.csv")
|
||||
)
|
||||
|
||||
assert sample_submission_files, "Error: No sample submission file found in {% include "scenarios.data_science.share:scen.input_path" %}"
|
||||
|
||||
# Use first matching file
|
||||
sample_submission_name = sample_submission_files[0].name
|
||||
SAMPLE_SUBMISSION_PATH = str(sample_submission_files[0])
|
||||
print(f"Using sample submission file: {sample_submission_name}")
|
||||
|
||||
# Check if the sample submission file exists
|
||||
assert Path(SAMPLE_SUBMISSION_PATH).exists(), f"Error: {sample_submission_name} not found at {SAMPLE_SUBMISSION_PATH}"
|
||||
|
||||
# Check if our submission file exists
|
||||
assert Path("submission.csv").exists(), "Error: submission.csv not found"
|
||||
|
||||
sample_submission = pd.read_csv(SAMPLE_SUBMISSION_PATH)
|
||||
our_submission = pd.read_csv("submission.csv")
|
||||
|
||||
success = True
|
||||
# Print the columns of the sample submission file
|
||||
print(f"Columns in {sample_submission_name}:", sample_submission.columns)
|
||||
print("Columns in our_submission.csv:", our_submission.columns)
|
||||
|
||||
for col in sample_submission.columns:
|
||||
if col not in our_submission.columns:
|
||||
success = False
|
||||
print(f"Column {col} not found in submission.csv")
|
||||
|
||||
if success:
|
||||
print(f"submission.csv's columns aligns with {sample_submission_name} .")
|
||||
else:
|
||||
raise AssertionError(f"submission.csv's columns does not align with {sample_submission_name} .")
|
||||
|
||||
|
||||
# Print the first 5 rows of the two submission files, with columns separated by commas.
|
||||
def print_first_rows(file_path, file_name, num_rows=5):
|
||||
print(f"\nFirst {num_rows} rows of {file_name}:")
|
||||
try:
|
||||
with open(file_path, "r") as file:
|
||||
for i, line in enumerate(file):
|
||||
if i < num_rows:
|
||||
print(line.strip())
|
||||
else:
|
||||
break
|
||||
except FileNotFoundError:
|
||||
print(f"Error: {file_name} not found.")
|
||||
|
||||
|
||||
print_first_rows(SAMPLE_SUBMISSION_PATH, sample_submission_name)
|
||||
print_first_rows("submission.csv", "submission.csv")
|
||||
|
||||
assert calculate_md5("scores.csv") == file_md5, "scores.csv should not be rewritten"
|
||||
print(
|
||||
f"\nPlease Checked the content of the submission file(submission.csv should has the same format with {sample_submission_name} but might not the same index with {sample_submission_name}). "
|
||||
)
|
||||
@@ -0,0 +1,7 @@
|
||||
from rdagent.components.coder.CoSTEER.task import CoSTEERTask
|
||||
|
||||
|
||||
# Because we use isinstance to distinguish between different types of tasks, we need to use sub classes to represent different types of tasks
|
||||
class PipelineTask(CoSTEERTask):
|
||||
def __init__(self, name: str = "Pipeline", *args, **kwargs) -> None:
|
||||
super().__init__(name=name, *args, **kwargs)
|
||||
@@ -0,0 +1,197 @@
|
||||
pipeline_coder:
|
||||
system: |-
|
||||
You are a world-class data scientist and machine learning engineer with deep expertise in statistics, mathematics, and computer science.
|
||||
Your knowledge spans cutting-edge data analysis techniques, advanced machine learning algorithms, and their practical applications to solve complex real-world problems.
|
||||
|
||||
## Task Description
|
||||
{{ task_desc }}
|
||||
|
||||
## The runtime environment your code will running on
|
||||
{{ runtime_environment }}
|
||||
|
||||
## Specification your code should follow
|
||||
{{ spec }}
|
||||
|
||||
{% if queried_similar_successful_knowledge|length != 0 or queried_former_failed_knowledge|length != 0 %}
|
||||
## Relevant Information for This Task
|
||||
{% endif %}
|
||||
|
||||
{% if queried_similar_successful_knowledge|length != 0 %}
|
||||
--------- Successful Implementations for Similar Models ---------
|
||||
====={% for similar_successful_knowledge in queried_similar_successful_knowledge %} Model {{ loop.index }}:=====
|
||||
{{ similar_successful_knowledge.target_task.get_task_information() }}
|
||||
=====Code:=====
|
||||
{{ similar_successful_knowledge.implementation.all_codes }}
|
||||
{% endfor %}
|
||||
{% endif %}
|
||||
|
||||
{% if queried_former_failed_knowledge|length != 0 %}
|
||||
--------- Previous Failed Attempts ---------
|
||||
{% for former_failed_knowledge in queried_former_failed_knowledge %} Attempt {{ loop.index }}:
|
||||
=====Code:=====
|
||||
{{ former_failed_knowledge.implementation.all_codes }}
|
||||
=====Feedback:=====
|
||||
{{ former_failed_knowledge.feedback }}
|
||||
{% endfor %}
|
||||
{% endif %}
|
||||
|
||||
|
||||
## Guidelines
|
||||
1. Ensure that the dataset is loaded strictly from `{% include "scenarios.data_science.share:scen.input_path" %}`, following the exact folder structure described in the **Data Folder Description**, and do not attempt to load data from the current directory (`./`).
|
||||
2. You should avoid using logging module to output information in your generated code, and instead use the print() function.
|
||||
|
||||
## Exploratory Data Analysis (EDA) part(Required):
|
||||
- Before returning the data, you should always add an EDA part describing the data to help the following steps understand the data better.
|
||||
- The EDA part should include but not limited in the following information in plain text:
|
||||
- The shape of the data.
|
||||
- The first 5 rows of the data.
|
||||
- The data types of each column.
|
||||
- The number of missing values in each column.
|
||||
- The number of unique values in each column.
|
||||
- The distribution of the target variable.
|
||||
- Any other information that you think is important for the following steps.
|
||||
- The EDA part should be drafted in plain text sending to standard output with command print or other similar functions with no more than ten thousand characters in the following schema:
|
||||
=== Start of EDA part ===
|
||||
{ You EDA output content }
|
||||
=== End of EDA part ===
|
||||
User will use the following code to match: re.search(r"(.*?)=== Start of EDA part ===(.*)=== End of EDA part ===", stdout, re.DOTALL).groups()[1]
|
||||
- An evaluation agent will help to check whether the EDA part is added correctly.
|
||||
- During the EDA part, you should try to avoid any irrelevant information sending to the standard output.
|
||||
{% include "scenarios.data_science.share:guidelines.coding" %}
|
||||
|
||||
{% if enable_model_dump %}
|
||||
## Model Dumping
|
||||
{% include "components.coder.data_science.share.prompts:dump_model_coder.guideline" %}
|
||||
{% endif %}
|
||||
|
||||
|
||||
|
||||
## Output Format
|
||||
{% if out_spec %}
|
||||
{{ out_spec }}
|
||||
{% else %}
|
||||
Please response the code in the following json format. Here is an example structure for the JSON output:
|
||||
{
|
||||
"code": "The Python code as a string."
|
||||
}
|
||||
{% endif %}
|
||||
|
||||
user: |-
|
||||
--------- Competition Information ---------
|
||||
{{ competition_info }}
|
||||
|
||||
--------- Data Folder Description (All path are relative to the data folder) ---------
|
||||
{{ folder_spec }}
|
||||
|
||||
{% if latest_code %}
|
||||
--------- Former code ---------
|
||||
{{ latest_code }}
|
||||
{% if latest_code_feedback is not none %}
|
||||
--------- Feedback to former code ---------
|
||||
{{ latest_code_feedback }}
|
||||
The former code contains errors. You should correct the code based on the provided information, ensuring you do not repeat the same mistakes.
|
||||
{% else %}
|
||||
The former code is correct. You should try to improve the code based on the provided task while not changing the irrelevant parts.
|
||||
{% endif %}
|
||||
{% endif %}
|
||||
|
||||
You should strictly follow the code specifications provided by the specification to implement the function.
|
||||
|
||||
user_v3: |-
|
||||
--------- Competition Information ---------
|
||||
{{ competition_info }}
|
||||
|
||||
--------- Data Folder Description (All path are relative to the data folder) ---------
|
||||
{{ folder_spec }}
|
||||
|
||||
{% if latest_code %}
|
||||
--------- Former code ---------
|
||||
{{ latest_code }}
|
||||
{% if latest_code_feedback is not none %}
|
||||
--------- Feedback to former code ---------
|
||||
{{ latest_code_feedback }}
|
||||
The former code contains errors. You should correct the code based on the provided information, ensuring you do not repeat the same mistakes.
|
||||
Keep the part that already seem correct intact. Avoid modifying them to refrain from introducing new errors.
|
||||
{% else %}
|
||||
The former code is correct. You should try to improve the code based on the provided task while not changing the irrelevant parts.
|
||||
{% endif %}
|
||||
{% endif %}
|
||||
|
||||
You should strictly follow the code specifications provided by the specification to implement the function.
|
||||
|
||||
pipeline_eval:
|
||||
system: |-
|
||||
You are a data scientist responsible for evaluating code generation.
|
||||
|
||||
## Task Description
|
||||
The user is trying to build a code in the following scenario:
|
||||
{{ scenario }}
|
||||
|
||||
The main code generation task is as follows:
|
||||
{{ task_desc }}
|
||||
|
||||
The details on how to structure the code are given in the specification:
|
||||
{{ spec }}
|
||||
|
||||
{% if is_sub_enabled %}
|
||||
## Evaluation Scope
|
||||
Your focus is to check whether the workflow code:
|
||||
Step 1: Executes successfully without any errors. Please distinguish between the errors and warnings.
|
||||
|
||||
Step 2: Correctly generates a final submission in the correct format, ensuring: they align with the submission structure, the index names and column names should match the sample, and the items should not be empty or apparently incorrect.
|
||||
|
||||
Step 3: Aligns with the competition requirements. This includes:
|
||||
- CAREFULLY ANALYZE WHETHER THE EXPERIMENTAL SETUP AND CODE MAY CAUSE MISALIGNMENT BETWEEN VALIDATION AND TEST PERFORMANCE.
|
||||
- Confirm strict adherence to the competition's evaluation rules listed in `scenario`:
|
||||
- Exact match between the implementation code of metric and the requirements of the scenario. The metric number is not the focus.
|
||||
- Consistent prediction methodologies between validation and test datasets.
|
||||
- No shortcuts or fold-specific strategies applied inconsistently.
|
||||
- Rigorous checks for corner-case consistency.
|
||||
- If such discrepancies or risks are found:
|
||||
- Clearly document these issues in `code`.
|
||||
- Begin your `code` with `[Evaluation error]`, explicitly stating the evaluation alignment issues causing experiment failure.
|
||||
- If no issues are found, begin your `code` with `[Code analysis]`, providing a detailed analysis of the code quality, readability, and adherence to specifications.
|
||||
|
||||
## Evaluation Criteria
|
||||
You will be given the execution output (`stdout`) to determine correctness.
|
||||
|
||||
[Note]
|
||||
1. Model performance is NOT a concern in this evaluation—only correct execution and formatting matter.
|
||||
2. You only check the format of the submission since we only feed you part of the data, so the submission might has different index to the sample submission data.
|
||||
|
||||
Please respond with your feedback in the following JSON format and order
|
||||
```json
|
||||
{
|
||||
"execution": "Describe whether the code executed successfully, correctly integrating all components and generating the final submission. Include any errors or issues encountered, and append all error messages and full traceback details without summarizing or omitting any information.",
|
||||
"return_checking": "Verify the generated files, particularly the submission file. Ensure that its format matches the sample submission, checking the index, column names, and CSV content.",
|
||||
"code": "Begin explicitly with [Code analysis] or [Evaluation error]. Provide feedback on code quality, readability, adherence to the given specifications, and alignment with competition requirements.",
|
||||
"final_decision": <true/false>
|
||||
}
|
||||
```
|
||||
{% else %}
|
||||
## Evaluation Scope
|
||||
Your focus is to check whether the workflow code executes successfully.
|
||||
|
||||
You will be given the execution output (`stdout`) to determine correctness.
|
||||
|
||||
[Note]
|
||||
1. Model performance is NOT a concern in this evaluation—only correct execution and formatting matter.
|
||||
|
||||
Please respond with your feedback in the following JSON format and order
|
||||
```json
|
||||
{
|
||||
"execution": "Describe whether the code executed successfully. Include any errors or issues encountered, and append all error messages and full traceback details without summarizing or omitting any information.",
|
||||
"return_checking": "Describe the expected file to be generated.",
|
||||
"code": "Provide feedback on code quality, readability, and adherence to the given specifications.",
|
||||
"final_decision": <true/false>
|
||||
}
|
||||
```
|
||||
{% endif %}
|
||||
# NOTE: when is_sub_enabled == False, we don't have any checking about the return. So it is just placeholder currently
|
||||
|
||||
user: |-
|
||||
--------- code generated by user ---------
|
||||
{{ code }}
|
||||
|
||||
--------- code running stdout ---------
|
||||
{{ stdout }}
|
||||
@@ -0,0 +1,15 @@
|
||||
# CoSTEER
|
||||
|
||||
- subworkspace使用主experiment_workspace `RD-Agent/rdagent/scenarios/data_science/experiment/experiment.py`
|
||||
|
||||
## evolving_strategy ( implement_one_task() )
|
||||
|
||||
1. xxxTask (in exp.py)
|
||||
- spec
|
||||
- description
|
||||
2.
|
||||
|
||||
## evaluator
|
||||
|
||||
1. queried_knowledge部分 共用
|
||||
2. eval_test脚本
|
||||
@@ -0,0 +1,245 @@
|
||||
"""
|
||||
|
||||
Loop should not large change exclude
|
||||
- Action Choice[current data loader & spec]
|
||||
- other should share
|
||||
- Propose[choice] => Task[Choice] => CoSTEER =>
|
||||
-
|
||||
|
||||
Extra feature:
|
||||
- cache
|
||||
|
||||
|
||||
File structure
|
||||
- ___init__.py: the entrance/agent of coder
|
||||
- evaluator.py
|
||||
- conf.py
|
||||
- exp.py: everything under the experiment, e.g.
|
||||
- Task
|
||||
- Experiment
|
||||
- Workspace
|
||||
- test.py
|
||||
- Each coder could be tested.
|
||||
"""
|
||||
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Dict
|
||||
|
||||
from rdagent.app.data_science.conf import DS_RD_SETTING
|
||||
from rdagent.components.coder.CoSTEER import CoSTEER
|
||||
from rdagent.components.coder.CoSTEER.evaluators import (
|
||||
CoSTEERMultiEvaluator,
|
||||
CoSTEERSingleFeedback,
|
||||
)
|
||||
from rdagent.components.coder.CoSTEER.evolving_strategy import (
|
||||
MultiProcessEvolvingStrategy,
|
||||
)
|
||||
from rdagent.components.coder.CoSTEER.knowledge_management import (
|
||||
CoSTEERQueriedKnowledge,
|
||||
)
|
||||
from rdagent.components.coder.data_science.conf import (
|
||||
DSCoderCoSTEERSettings,
|
||||
get_ds_env,
|
||||
)
|
||||
from rdagent.components.coder.data_science.raw_data_loader.eval import (
|
||||
DataLoaderCoSTEEREvaluator,
|
||||
)
|
||||
from rdagent.components.coder.data_science.raw_data_loader.exp import DataLoaderTask
|
||||
from rdagent.core.exception import CoderError
|
||||
from rdagent.core.experiment import FBWorkspace
|
||||
from rdagent.core.scenario import Scenario
|
||||
from rdagent.oai.llm_utils import APIBackend
|
||||
from rdagent.utils.agent.ret import PythonAgentOut
|
||||
from rdagent.utils.agent.tpl import T
|
||||
|
||||
DIRNAME = Path(__file__).absolute().resolve().parent
|
||||
|
||||
|
||||
class DataLoaderMultiProcessEvolvingStrategy(MultiProcessEvolvingStrategy):
|
||||
def implement_one_task(
|
||||
self,
|
||||
target_task: DataLoaderTask,
|
||||
queried_knowledge: CoSTEERQueriedKnowledge | None = None,
|
||||
workspace: FBWorkspace | None = None,
|
||||
prev_task_feedback: CoSTEERSingleFeedback | None = None,
|
||||
) -> dict[str, str]:
|
||||
# return a workspace with "load_data.py", "spec/load_data.md" inside
|
||||
# assign the implemented code to the new workspace.
|
||||
competition_info = self.scen.get_scenario_all_desc(eda_output=workspace.file_dict.get("EDA.md", None))
|
||||
runtime_environment = self.scen.get_runtime_environment()
|
||||
data_folder_info = self.scen.processed_data_folder_description
|
||||
data_loader_task_info = target_task.get_task_information()
|
||||
|
||||
queried_similar_successful_knowledge = (
|
||||
queried_knowledge.task_to_similar_task_successful_knowledge[data_loader_task_info]
|
||||
if queried_knowledge is not None
|
||||
else []
|
||||
)
|
||||
queried_former_failed_knowledge = (
|
||||
queried_knowledge.task_to_former_failed_traces[data_loader_task_info]
|
||||
if queried_knowledge is not None
|
||||
else []
|
||||
)
|
||||
queried_former_failed_knowledge = (
|
||||
[
|
||||
knowledge
|
||||
for knowledge in queried_former_failed_knowledge[0]
|
||||
if knowledge.implementation.file_dict.get("load_data.py") != workspace.file_dict.get("load_data.py")
|
||||
],
|
||||
queried_former_failed_knowledge[1],
|
||||
)
|
||||
|
||||
# 1. specifications
|
||||
# TODO: We may move spec into a separated COSTEER task
|
||||
if DS_RD_SETTING.spec_enabled:
|
||||
if "spec/data_loader.md" not in workspace.file_dict: # Only generate the spec once
|
||||
system_prompt = T(".prompts:spec.system").r(
|
||||
runtime_environment=runtime_environment,
|
||||
task_desc=data_loader_task_info,
|
||||
competition_info=competition_info,
|
||||
folder_spec=data_folder_info,
|
||||
)
|
||||
data_loader_prompt = T(".prompts:spec.user.data_loader").r(
|
||||
latest_spec=workspace.file_dict.get("spec/data_loader.md")
|
||||
)
|
||||
feature_prompt = T(".prompts:spec.user.feature").r(
|
||||
latest_spec=workspace.file_dict.get("spec/feature.md")
|
||||
)
|
||||
model_prompt = T(".prompts:spec.user.model").r(latest_spec=workspace.file_dict.get("spec/model.md"))
|
||||
ensemble_prompt = T(".prompts:spec.user.ensemble").r(
|
||||
latest_spec=workspace.file_dict.get("spec/ensemble.md")
|
||||
)
|
||||
workflow_prompt = T(".prompts:spec.user.workflow").r(
|
||||
latest_spec=workspace.file_dict.get("spec/workflow.md")
|
||||
)
|
||||
|
||||
spec_session = APIBackend().build_chat_session(session_system_prompt=system_prompt)
|
||||
|
||||
data_loader_spec = spec_session.build_chat_completion(user_prompt=data_loader_prompt)
|
||||
feature_spec = spec_session.build_chat_completion(user_prompt=feature_prompt)
|
||||
model_spec = spec_session.build_chat_completion(user_prompt=model_prompt)
|
||||
ensemble_spec = spec_session.build_chat_completion(user_prompt=ensemble_prompt)
|
||||
workflow_spec = spec_session.build_chat_completion(user_prompt=workflow_prompt)
|
||||
else:
|
||||
data_loader_spec = workspace.file_dict["spec/data_loader.md"]
|
||||
feature_spec = workspace.file_dict["spec/feature.md"]
|
||||
model_spec = workspace.file_dict["spec/model.md"]
|
||||
ensemble_spec = workspace.file_dict["spec/ensemble.md"]
|
||||
workflow_spec = workspace.file_dict["spec/workflow.md"]
|
||||
|
||||
# 2. code
|
||||
system_prompt = T(".prompts:data_loader_coder.system").r(
|
||||
task_desc=data_loader_task_info,
|
||||
queried_similar_successful_knowledge=queried_similar_successful_knowledge,
|
||||
queried_former_failed_knowledge=queried_former_failed_knowledge[0],
|
||||
out_spec=PythonAgentOut.get_spec(),
|
||||
)
|
||||
code_spec = (
|
||||
data_loader_spec
|
||||
if DS_RD_SETTING.spec_enabled
|
||||
else T("scenarios.data_science.share:component_spec.general").r(
|
||||
spec=T("scenarios.data_science.share:component_spec.DataLoadSpec").r(),
|
||||
test_code=(DIRNAME / "eval_tests" / "data_loader_test.txt").read_text(),
|
||||
)
|
||||
)
|
||||
user_prompt = T(".prompts:data_loader_coder.user").r(
|
||||
competition_info=competition_info,
|
||||
code_spec=code_spec,
|
||||
folder_spec=data_folder_info,
|
||||
latest_code=workspace.file_dict.get("load_data.py"),
|
||||
latest_code_feedback=prev_task_feedback,
|
||||
)
|
||||
|
||||
for _ in range(5):
|
||||
data_loader_code = PythonAgentOut.extract_output(
|
||||
APIBackend().build_messages_and_create_chat_completion(
|
||||
user_prompt=user_prompt,
|
||||
system_prompt=system_prompt,
|
||||
)
|
||||
)
|
||||
if data_loader_code != workspace.file_dict.get("load_data.py"):
|
||||
break
|
||||
else:
|
||||
user_prompt = user_prompt + "\nPlease avoid generating same code to former code!"
|
||||
else:
|
||||
raise CoderError("Failed to generate a new data loader code.")
|
||||
|
||||
return (
|
||||
{
|
||||
"spec/data_loader.md": data_loader_spec,
|
||||
"spec/feature.md": feature_spec,
|
||||
"spec/model.md": model_spec,
|
||||
"spec/ensemble.md": ensemble_spec,
|
||||
"spec/workflow.md": workflow_spec,
|
||||
"load_data.py": data_loader_code,
|
||||
}
|
||||
if DS_RD_SETTING.spec_enabled
|
||||
else {
|
||||
"load_data.py": data_loader_code,
|
||||
}
|
||||
)
|
||||
|
||||
def assign_code_list_to_evo(self, code_list: list[dict[str, str]], evo):
|
||||
"""
|
||||
Assign the code list to the evolving item.
|
||||
|
||||
The code list is aligned with the evolving item's sub-tasks.
|
||||
If a task is not implemented, put a None in the list.
|
||||
"""
|
||||
for index in range(len(evo.sub_tasks)):
|
||||
if code_list[index] is None:
|
||||
continue
|
||||
if evo.sub_workspace_list[index] is None:
|
||||
# evo.sub_workspace_list[index] = FBWorkspace(target_task=evo.sub_tasks[index])
|
||||
evo.sub_workspace_list[index] = evo.experiment_workspace
|
||||
evo.sub_workspace_list[index].inject_files(**code_list[index])
|
||||
return evo
|
||||
|
||||
|
||||
class DataLoaderCoSTEER(CoSTEER):
|
||||
def __init__(
|
||||
self,
|
||||
scen: Scenario,
|
||||
*args,
|
||||
**kwargs,
|
||||
) -> None:
|
||||
settings = DSCoderCoSTEERSettings()
|
||||
eva = CoSTEERMultiEvaluator(
|
||||
DataLoaderCoSTEEREvaluator(scen=scen), scen=scen
|
||||
) # Please specify whether you agree running your eva in parallel or not
|
||||
es = DataLoaderMultiProcessEvolvingStrategy(scen=scen, settings=settings)
|
||||
|
||||
super().__init__(
|
||||
*args,
|
||||
settings=settings,
|
||||
eva=eva,
|
||||
es=es,
|
||||
evolving_version=2,
|
||||
scen=scen,
|
||||
max_loop=DS_RD_SETTING.coder_max_loop,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
def develop(self, exp):
|
||||
new_exp = super().develop(exp)
|
||||
|
||||
env = get_ds_env(
|
||||
extra_volumes={
|
||||
f"{DS_RD_SETTING.local_data_path}/{self.scen.competition}": T(
|
||||
"scenarios.data_science.share:scen.input_path"
|
||||
).r()
|
||||
},
|
||||
running_timeout_period=DS_RD_SETTING.full_timeout,
|
||||
)
|
||||
|
||||
stdout = new_exp.experiment_workspace.execute(env=env, entry=f"python test/data_loader_test.py")
|
||||
match = re.search(r"(.*?)=== Start of EDA part ===(.*)=== End of EDA part ===", stdout, re.DOTALL)
|
||||
eda_output = match.groups()[1] if match else None
|
||||
if eda_output is not None:
|
||||
new_exp.experiment_workspace.inject_files(**{"EDA.md": eda_output})
|
||||
else:
|
||||
eda_output = "No EDA output."
|
||||
new_exp.experiment_workspace.inject_files(**{"EDA.md": eda_output})
|
||||
return new_exp
|
||||
@@ -0,0 +1,89 @@
|
||||
# tess successfully running.
|
||||
# (GPT) if it aligns with the spec & rationality of the spec.
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
from rdagent.app.data_science.conf import DS_RD_SETTING
|
||||
from rdagent.components.coder.CoSTEER.evaluators import (
|
||||
CoSTEEREvaluator,
|
||||
CoSTEERSingleFeedback,
|
||||
)
|
||||
from rdagent.components.coder.CoSTEER.knowledge_management import (
|
||||
CoSTEERQueriedKnowledgeV2,
|
||||
)
|
||||
from rdagent.components.coder.data_science.conf import get_ds_env
|
||||
from rdagent.components.coder.data_science.utils import remove_eda_part
|
||||
from rdagent.core.experiment import FBWorkspace, Task
|
||||
from rdagent.utils.agent.tpl import T
|
||||
from rdagent.utils.agent.workflow import build_cls_from_json_with_retry
|
||||
|
||||
DIRNAME = Path(__file__).absolute().resolve().parent
|
||||
|
||||
DataLoaderEvalFeedback = CoSTEERSingleFeedback
|
||||
|
||||
|
||||
class DataLoaderCoSTEEREvaluator(CoSTEEREvaluator):
|
||||
def evaluate(
|
||||
self,
|
||||
target_task: Task,
|
||||
implementation: FBWorkspace,
|
||||
gt_implementation: FBWorkspace,
|
||||
queried_knowledge: CoSTEERQueriedKnowledgeV2 = None,
|
||||
**kwargs,
|
||||
) -> DataLoaderEvalFeedback:
|
||||
target_task_information = target_task.get_task_information()
|
||||
if (
|
||||
queried_knowledge is not None
|
||||
and target_task_information in queried_knowledge.success_task_to_knowledge_dict
|
||||
):
|
||||
return queried_knowledge.success_task_to_knowledge_dict[target_task_information].feedback
|
||||
elif queried_knowledge is not None and target_task_information in queried_knowledge.failed_task_info_set:
|
||||
return DataLoaderEvalFeedback(
|
||||
execution="This task has failed too many times, skip implementation.",
|
||||
return_checking="This task has failed too many times, skip implementation.",
|
||||
code="This task has failed too many times, skip implementation.",
|
||||
final_decision=False,
|
||||
)
|
||||
|
||||
env = get_ds_env(extra_volumes={self.scen.debug_path: T("scenarios.data_science.share:scen.input_path").r()})
|
||||
|
||||
# TODO: do we need to clean the generated temporary content?
|
||||
fname = "test/data_loader_test.py"
|
||||
test_code = (DIRNAME / "eval_tests" / "data_loader_test.txt").read_text()
|
||||
implementation.inject_files(**{fname: test_code})
|
||||
stdout, ret_code = implementation.execute_ret_code(env=env, entry=f"python {fname}")
|
||||
match = re.search(r"(.*?)=== Start of EDA part ===(.*)=== End of EDA part ===(.*)", stdout, re.DOTALL)
|
||||
stdout_part_1, eda_output, stdout_part_2 = match.groups() if match else (stdout, None, "")
|
||||
stdout = stdout_part_1 + stdout_part_2
|
||||
if eda_output is not None and len(eda_output.split(" ")) > 10000:
|
||||
eda_output += "Length of EDA output is too long, truncated. Please reject this implementation and motivate it to reduce the length of EDA output."
|
||||
|
||||
if "main.py" in implementation.file_dict and ret_code == 0:
|
||||
workflow_stdout = implementation.execute(env=env, entry="python main.py")
|
||||
workflow_stdout = remove_eda_part(workflow_stdout)
|
||||
else:
|
||||
workflow_stdout = None
|
||||
|
||||
system_prompt = T(".prompts:data_loader_eval.system").r(
|
||||
task_desc=target_task.get_task_information(),
|
||||
test_code=test_code,
|
||||
code=implementation.file_dict["load_data.py"],
|
||||
workflow_stdout=workflow_stdout,
|
||||
workflow_code=implementation.all_codes,
|
||||
)
|
||||
user_prompt = T(".prompts:data_loader_eval.user").r(
|
||||
stdout=stdout,
|
||||
eda_output=eda_output,
|
||||
workflow_stdout=workflow_stdout,
|
||||
)
|
||||
|
||||
fb = build_cls_from_json_with_retry(
|
||||
DataLoaderEvalFeedback,
|
||||
system_prompt=system_prompt,
|
||||
user_prompt=user_prompt,
|
||||
init_kwargs_update_func=DataLoaderEvalFeedback.val_and_update_init_dict,
|
||||
)
|
||||
fb.final_decision = fb.final_decision and ret_code == 0
|
||||
|
||||
return fb
|
||||
@@ -0,0 +1,83 @@
|
||||
"""
|
||||
Tests for `load_data` in load_data.py
|
||||
"""
|
||||
|
||||
import pickle
|
||||
|
||||
import pandas as pd
|
||||
from load_data import load_data
|
||||
|
||||
import sys
|
||||
import reprlib
|
||||
from joblib.memory import MemorizedFunc
|
||||
|
||||
|
||||
def get_original_code(func):
|
||||
if isinstance(func, MemorizedFunc):
|
||||
return func.func.__code__
|
||||
return func.__code__
|
||||
|
||||
|
||||
def debug_info_print(func):
|
||||
aRepr = reprlib.Repr()
|
||||
aRepr.maxother=300
|
||||
def wrapper(*args, **kwargs):
|
||||
original_code = get_original_code(func)
|
||||
def local_trace(frame, event, arg):
|
||||
if event == "return" and frame.f_code == original_code:
|
||||
print("\n" + "="*20 + "Running data_load code, local variable values:" + "="*20)
|
||||
for k, v in frame.f_locals.items():
|
||||
printed = aRepr.repr(v)
|
||||
print(f"{k}:\n {printed}")
|
||||
print("="*20 + "Local variable values end" + "="*20)
|
||||
return local_trace
|
||||
|
||||
sys.settrace(local_trace)
|
||||
try:
|
||||
return func(*args, **kwargs)
|
||||
finally:
|
||||
sys.settrace(None)
|
||||
return wrapper
|
||||
|
||||
X, y, X_test, test_ids = debug_info_print(load_data)()
|
||||
|
||||
|
||||
def get_length(data):
|
||||
return data.shape[0] if hasattr(data, 'shape') else len(data)
|
||||
|
||||
|
||||
def get_width(data):
|
||||
return data.shape[1:] if hasattr(data, 'shape') else 1
|
||||
|
||||
|
||||
def get_column_list(data):
|
||||
return data.columns.tolist() if isinstance(data, pd.DataFrame) else None
|
||||
|
||||
assert X is not None, "Training data (X) is None."
|
||||
assert y is not None, "Training labels (y) are None."
|
||||
assert X_test is not None, "Test data (X_test) is None."
|
||||
assert test_ids is not None, "Test IDs (test_ids) are None."
|
||||
|
||||
assert get_length(X_test) == get_length(
|
||||
test_ids
|
||||
), f"Mismatch in length of test images and test IDs: X_test ({get_length(X_test)}) and test_ids ({get_length(test_ids)})"
|
||||
assert get_length(X) == get_length(
|
||||
y
|
||||
), f"Mismatch in length of training images and labels: X ({get_length(X)}) and y ({get_length(y)})"
|
||||
|
||||
assert get_length(X) != 0, f"Training data is empty."
|
||||
assert get_length(y) != 0, f"Training labels are empty."
|
||||
assert get_length(X_test) != 0, f"Test data is empty."
|
||||
|
||||
assert get_width(X) == get_width(
|
||||
X_test
|
||||
), "Mismatch in width of training and test data. Width means the number of features."
|
||||
|
||||
if isinstance(X, pd.DataFrame) and isinstance(X_test, pd.DataFrame):
|
||||
assert get_column_list(X) == get_column_list(X_test), "Mismatch in column names of training and test data."
|
||||
|
||||
assert get_width(X) == get_width(
|
||||
X_test
|
||||
), "Mismatch in width of training and test data. Width means the number of features."
|
||||
|
||||
print("Data loader test passed successfully. Length of test images matches length of test IDs.")
|
||||
@@ -0,0 +1,6 @@
|
||||
from rdagent.components.coder.CoSTEER.task import CoSTEERTask
|
||||
|
||||
|
||||
# Because we use isinstance to distinguish between different types of tasks, we need to use sub classes to represent different types of tasks
|
||||
class DataLoaderTask(CoSTEERTask):
|
||||
pass
|
||||
@@ -0,0 +1,402 @@
|
||||
|
||||
spec:
|
||||
system: |-
|
||||
You are a world-class data scientist and machine learning engineer with deep expertise in statistics, mathematics, and computer science.
|
||||
Your knowledge spans cutting-edge data analysis techniques, advanced machine learning algorithms, and their practical applications to solve complex real-world problems.
|
||||
|
||||
Currently, you are working on a Kaggle competition project.
|
||||
This project involves analyzing data and building models to beat other competitors, with the code being generated by large language models.
|
||||
|
||||
The runtime environment you are working in includes the following libraries and their respective versions:
|
||||
{{ runtime_environment }}
|
||||
|
||||
Your overall task is provided below:
|
||||
{{ task_desc }}
|
||||
|
||||
Your task is to write five specification texts (in markdown format) for the following tasks, based on the competition information provided
|
||||
- Data loading (and preprocessing)
|
||||
- Feature Engineering
|
||||
- Model Building
|
||||
- Ensemble
|
||||
- The overall workflow
|
||||
|
||||
The specifications for each step should be tailored to the competition information provided.
|
||||
|
||||
Your specification should consists two parts:
|
||||
1. The function definition in code format, including type annotations and a clear, complete docstring that describes the function's purpose, input parameters, return value, and any relevant exceptions.
|
||||
2. Additional information or notes that the coder should consider while implementing the function.
|
||||
|
||||
Your specifications should include only the function definition and docstring, without any code implementation or inline comments.
|
||||
|
||||
## Competition Information for This Task
|
||||
{{ competition_info }}
|
||||
|
||||
----------- Folder Description (All path are relative to the data folder) ---------
|
||||
- Ensure that all columns in sample_submission can be generated.
|
||||
{{ folder_spec }}
|
||||
|
||||
user:
|
||||
data_loader: |-
|
||||
Data loader specification text should follow these detailed requirements:
|
||||
1. Function Interface:
|
||||
- Function Name: `load_data`
|
||||
- Input: No input arguments.
|
||||
- Output:
|
||||
- `X` (DT, define based on competition information): Feature matrix for training data.
|
||||
- `y` (DT): Target vector for training data.
|
||||
- `X_test` (DT): Feature matrix for test data.
|
||||
- `test_ids` (DT): Identifiers for the test data.
|
||||
- Docstring Requirements:
|
||||
- Describe the purpose of the function.
|
||||
- Specify the data source location (`{% include "scenarios.data_science.share:scen.input_path" %}`).
|
||||
- Clearly define the structure and type of the output.
|
||||
- Inferred data shape to each input and output data variables. To uncertain dimension, use -1.
|
||||
2. Notes:
|
||||
- Update `DT` (data type) based on the specific competition dataset. This can include `pd.DataFrame`, `np.array`, `torch.Tensor`, etc.
|
||||
- Only set the DT of variables without inferring the shape of these variables since you don't know the shape of the data.
|
||||
|
||||
Responsibilities and notes of an implemented data loader that aligns with the generated specification.
|
||||
{% include "scenarios.data_science.share:component_spec.DataLoadSpec" %}
|
||||
|
||||
{% if latest_spec %}
|
||||
6. Former Specification:
|
||||
{{ latest_spec }}
|
||||
You should follow the provided specifications to improve this task.
|
||||
{% endif %}
|
||||
|
||||
## Output Format
|
||||
You should return the specification in markdown format directly, while the **function definition** within it should be in code format, tailored to the Competition Information, with detailed explanations provided in the docstring.
|
||||
|
||||
feature: |-
|
||||
Feature engineering specification text should adhere to the following requirements:
|
||||
1. Function Interface:
|
||||
- Function Name: `feat_eng`
|
||||
- Parameters:
|
||||
- `X` (DT): Train data to be transformed.
|
||||
- `y` (DT): Train label data.
|
||||
- `X_test` (DT): Test data.
|
||||
- Output:
|
||||
- `X_transformed` (DT): Transformed train data.
|
||||
- `y_transformed` (DT): Transformed train label data.
|
||||
- `X_test_transformed` (DT): Transformed test data.
|
||||
- Docstring Requirements:
|
||||
- Describe the purpose of the function.
|
||||
- Clarify the input parameters and their data types.
|
||||
- Define the structure and format of the output.
|
||||
- Inferred data shape to each input and output data variables. To uncertain dimension, use -1.
|
||||
|
||||
2. Precautions for Feature Engineering:
|
||||
- Well handle the shape of the data:
|
||||
- The sample size of the train data and the test data should be the same in all scenarios.
|
||||
- To some tabular or time-series data, you may add or remove some columns so your inferred column number may be unsure.
|
||||
- For scenarios where each dimension does not have a special meaning (like image, audio, and so on), the input shape and the output shape should be exactly the same in most cases unless there is a compelling reason to change them.
|
||||
- Integration with the Model Pipeline:
|
||||
- If feature engineering is deferred to the model pipeline for better overall performance, state explicitly that it will be handled at the model stage.
|
||||
- Model-related operations should not be implemented in this step. (e.g., it uses tools combined with models like torch.Dataset with rich data transformation/augmentation)
|
||||
- Otherwise, ensure this function applies all required transformations while avoiding data leakage.
|
||||
- General Considerations:
|
||||
- Ensure scalability for large datasets.
|
||||
- Handle missing values and outliers appropriately (e.g., impute, remove, or replace).
|
||||
- Ensure consistency between feature data types and transformations.
|
||||
- Prevent data leakage: Do not use information derived from the test set when transforming training data.
|
||||
- Domain-Specific Features:
|
||||
- Apply logic for competition-specific features (e.g., text vectorization, image augmentations, categorical encoding).
|
||||
|
||||
3. Code Standards:
|
||||
- Avoid using progress bars (e.g., `tqdm`) in the implementation.
|
||||
|
||||
4. Notes:
|
||||
- Align `DT` (data type) definitions with those in the Data Loader specification.
|
||||
- GPU and multiprocessing are available and are encouraged to use for accelerating transformations.
|
||||
- Only set the DT of variables without inferring the shape of these variables since you don't know the shape of the data.
|
||||
|
||||
{% if latest_spec %}
|
||||
5. Former Specification:
|
||||
{{ latest_spec }}
|
||||
You should follow the provided specifications to improve this task.
|
||||
{% endif %}
|
||||
|
||||
## Output Format
|
||||
You should return the specification in markdown format directly, while the **function definition** within it should be in code format, tailored to the Competition Information, with detailed explanations provided in the docstring.
|
||||
|
||||
model: |-
|
||||
Model building specification text should adhere to the following requirements:
|
||||
|
||||
1. Function Interface:
|
||||
- Function Name: `model_workflow`
|
||||
- Parameters:
|
||||
- `X` (DT): Training feature data.
|
||||
- `y` (DT): Training label data.
|
||||
- `val_X` (Optional[DT]): Validation feature data.
|
||||
- `val_y` (Optional[DT]): Validation label data.
|
||||
- `test_X` (Optional[DT]): Test feature data.
|
||||
- `hyper_params` (dict): Dictionary of hyperparameters for model configuration.
|
||||
- Output:
|
||||
- `pred_val` (Optional[DT]): Predictions on validation data.
|
||||
- `pred_test` (Optional[DT]): Predictions on test data.
|
||||
- `hyper_params` (dict): Updated dictionary of hyperparameters after training.
|
||||
- Docstring Requirements:
|
||||
- Describe the purpose of the function.
|
||||
- Clarify the input parameters and their data types.
|
||||
- Define the structure and format of the output.
|
||||
- Inferred data shape to each input and output data variables. To uncertain dimension, use -1.
|
||||
|
||||
2. Code Standards:
|
||||
- Do not use progress bars (e.g., `tqdm`) in the implementation.
|
||||
|
||||
3. Precautions:
|
||||
- Ensure input arrays (`X`, `y`, `val_X`, `val_y`, `test_X`) have consistent dimensions and shapes.
|
||||
- Use default values for hyperparameters if `hyper_params` is not provided.
|
||||
- Train the model on `X` and `y`.
|
||||
- Evaluate the model using `val_X` and `val_y` if validation data is available.
|
||||
- If `test_X` is provided, generate predictions for it.
|
||||
|
||||
4. Notes:
|
||||
- Align `DT` (data type) with the definitions used in Feature Engineering specifications.
|
||||
- The device has GPU support, so you are encouraged to use it for training if necessary to accelerate the process.
|
||||
- Some data transformations/augmentations can be included in this step (e.g., data tools provided by TensorFlow and Torch)
|
||||
|
||||
{% if latest_spec %}
|
||||
5. Former Specification:
|
||||
{{ latest_spec }}
|
||||
You should follow the provided specifications to improve this task.
|
||||
{% endif %}
|
||||
|
||||
## Output Format
|
||||
You should return the specification in markdown format directly, while the **function definition** within it should be in code format, tailored to the Competition Information, with detailed explanations provided in the docstring.
|
||||
|
||||
ensemble: |-
|
||||
Ensemble specification text adhere to the following requirements:
|
||||
1. Function Interface:
|
||||
- Function Name: `ensemble_workflow`
|
||||
- Parameters:
|
||||
- `test_preds_dict` (Dict[str, DT]): A dictionary of test predictions from different models. The key is the model file name.
|
||||
- `val_preds_dict` (Dict[str, DT]): A dictionary of validation predictions from different models. The key is the model file name.
|
||||
- `val_label` (DT): Validation label.
|
||||
- Output:
|
||||
- `final_pred` (DT): Ensemble prediction for the test data.
|
||||
- Docstring Requirements:
|
||||
- Describe the purpose of the function.
|
||||
- Clarify the input parameters and their data types.
|
||||
- Define the structure and format of the output.
|
||||
- Inferred data shape to each input and output data variables. To uncertain dimension, use -1.
|
||||
|
||||
2. Precautions:
|
||||
- Input Validation:
|
||||
- Ensure all predictions in `test_preds_dict` and `val_preds_dict` have consistent shapes and dimensions.
|
||||
- Verify that `val_label` is provided and matches the length of `val_preds_dict` predictions.
|
||||
- Handle empty or invalid inputs gracefully with appropriate error messages.
|
||||
- Metric Calculation and Storage:
|
||||
- Calculate the metric (mentioned in the evaluation section of the competition information) for each model and ensemble strategy on valid, and save the results in `scores.csv`, e.g.:
|
||||
```python
|
||||
scores = {}
|
||||
for model_name, val_pred in val_preds_dict.items():
|
||||
scores[model_name] = calculate_metric(val_label, val_pred)
|
||||
|
||||
...
|
||||
some code about ensemble strategy
|
||||
...
|
||||
ensemble_val_pred = ...
|
||||
|
||||
ensemble_score = calculate_metric(val_label, ensemble_val_pred)
|
||||
scores["ensemble"] = ensemble_score # Ensure "ensemble" is explicitly stored
|
||||
|
||||
scores_df = pd.DataFrame(scores.items(), columns=["Model", <metric_name>])
|
||||
scores_df.to_csv("scores.csv", index=False)
|
||||
```
|
||||
- Even if only one model is present, compute the ensemble score and store it under `"ensemble"`.
|
||||
|
||||
3. Code Standards:
|
||||
- Do not use progress bars (e.g., tqdm) in the code.
|
||||
|
||||
4. Notes:
|
||||
- Align `DT` (data type) definitions with those used in model specifications.
|
||||
- Ensure flexibility to handle multiple ensemble strategies based on competition requirements.
|
||||
- Only set the DT of variables without inferring the shape of these variables since you don't know the shape of the data.
|
||||
|
||||
{% if latest_spec %}
|
||||
5. Former Specification:
|
||||
{{ latest_spec }}
|
||||
You should follow the provided specifications to improve this task.
|
||||
{% endif %}
|
||||
|
||||
## Output Format
|
||||
You should return the specification in markdown format directly, while the **function definition** within it should be in code format, tailored to the Competition Information, with detailed explanations provided in the docstring.
|
||||
|
||||
workflow: |-
|
||||
{% include "scenarios.data_science.share:component_spec.Workflow" %}
|
||||
|
||||
{% if latest_spec %}
|
||||
7. Former Specification:
|
||||
{{ latest_spec }}
|
||||
You should follow the provided specifications to improve this task.
|
||||
{% endif %}
|
||||
|
||||
## Output Format
|
||||
You should return the specification in markdown format directly.
|
||||
You should create the rules based on the competition information instead of copying the requirements.
|
||||
|
||||
data_loader_coder:
|
||||
system: |-
|
||||
You are a world-class data scientist and machine learning engineer with deep expertise in statistics, mathematics, and computer science.
|
||||
Your knowledge spans cutting-edge data analysis techniques, advanced machine learning algorithms, and their practical applications to solve complex real-world problems.
|
||||
|
||||
## Task Description
|
||||
{{ task_desc }}
|
||||
|
||||
{% if queried_similar_successful_knowledge|length != 0 or queried_former_failed_knowledge|length != 0 %}
|
||||
## Relevant Information for This Task
|
||||
{% endif %}
|
||||
|
||||
{% if queried_similar_successful_knowledge|length != 0 %}
|
||||
--------- Successful Implementation Examples for Similar Task ---------
|
||||
====={% for similar_successful_knowledge in queried_similar_successful_knowledge %} Example {{ loop.index }}:=====
|
||||
{{ similar_successful_knowledge.target_task.get_task_information() }}
|
||||
=====Code:=====
|
||||
{{ similar_successful_knowledge.implementation.all_codes }}
|
||||
{% endfor %}
|
||||
{% endif %}
|
||||
|
||||
{% if queried_former_failed_knowledge|length != 0 %}
|
||||
--------- Previous Failed Attempts ---------
|
||||
{% for former_failed_knowledge in queried_former_failed_knowledge %} Attempt {{ loop.index }}:
|
||||
=====Code:=====
|
||||
{{ former_failed_knowledge.implementation.all_codes }}
|
||||
=====Feedback:=====
|
||||
{{ former_failed_knowledge.feedback }}
|
||||
{% endfor %}
|
||||
{% endif %}
|
||||
|
||||
## Guidelines
|
||||
1. Ensure that the dataset is loaded strictly from `{% include "scenarios.data_science.share:scen.input_path" %}`, following the exact folder structure described in the **Data Folder Description**, and do not attempt to load data from the current directory (`./`).
|
||||
2. You should avoid using logging module to output information in your generated code, and instead use the print() function.
|
||||
3. You should use the following cache decorator to cache the results of the function:
|
||||
```python
|
||||
from joblib import Memory
|
||||
memory = Memory(location='{% include "scenarios.data_science.share:scen.cache_path" %}', verbose=0)
|
||||
@memory.cache```
|
||||
{% include "scenarios.data_science.share:guidelines.coding" %}
|
||||
|
||||
## Exploratory Data Analysis (EDA) part(Required):
|
||||
- Before returning the data, you should always add an EDA part describing the data to help the following steps understand the data better.
|
||||
- The EDA part should include but not limited in the following information in plain text:
|
||||
- The shape of the data.
|
||||
- The first 5 rows of the data.
|
||||
- The data types of each column.
|
||||
- The number of missing values in each column.
|
||||
- The number of unique values in each column.
|
||||
- The distribution of the target variable.
|
||||
- Any other information that you think is important for the following steps.
|
||||
- The EDA part should be drafted in plain text sending to standard output with command print or other similar functions with no more than ten thousand characters in the following schema:
|
||||
=== Start of EDA part ===
|
||||
{ You EDA output content }
|
||||
=== End of EDA part ===
|
||||
User will use the following code to match: re.search(r"(.*?)=== Start of EDA part ===(.*)=== End of EDA part ===", stdout, re.DOTALL).groups()[1]
|
||||
- An evaluation agent will help to check whether the EDA part is added correctly.
|
||||
- During the EDA part, you should try to avoid any irrelevant information sending to the standard output.
|
||||
|
||||
## Output Format
|
||||
{% if out_spec %}
|
||||
{{ out_spec }}
|
||||
{% else %}
|
||||
Please response the code in the following json format. Here is an example structure for the JSON output:
|
||||
{
|
||||
"code": "The Python code as a string."
|
||||
}
|
||||
{% endif %}
|
||||
|
||||
user: |-
|
||||
--------- Competition Information ---------
|
||||
{{ competition_info }}
|
||||
|
||||
--------- Code Specification ---------
|
||||
{{ code_spec }}
|
||||
|
||||
--------- Data Folder Description (All path are relative to the data folder) ---------
|
||||
{{ folder_spec }}
|
||||
|
||||
{% if latest_code %}
|
||||
--------- Former code ---------
|
||||
{{ latest_code }}
|
||||
{% if latest_code_feedback is not none %}
|
||||
--------- Feedback to former code ---------
|
||||
{{ latest_code_feedback }}
|
||||
{% endif %}
|
||||
The former code contains errors. You should correct the code based on the provided information, ensuring you do not repeat the same mistakes.
|
||||
{% endif %}
|
||||
|
||||
You should strictly follow the code specifications provided by the specification to implement the function.
|
||||
|
||||
|
||||
data_loader_eval:
|
||||
system: |-
|
||||
You are a data scientist responsible for evaluating data loader code for a Kaggle-style machine learning competition project.
|
||||
|
||||
## Task Description
|
||||
{{ task_desc }}
|
||||
|
||||
## Data Loader Code
|
||||
The data loader code is located in `load_data.py`:
|
||||
```python
|
||||
{{ code }}
|
||||
```
|
||||
|
||||
## Testing Process
|
||||
The data loader is tested using the following script:
|
||||
```python
|
||||
{{ test_code }}
|
||||
```
|
||||
|
||||
{% if workflow_stdout is not none %}
|
||||
### Whole Workflow Consideration
|
||||
The data loader is part of the whole workflow. The user has executed the entire pipeline and provided additional stdout.
|
||||
|
||||
**Workflow Code:**
|
||||
{{ workflow_code }}
|
||||
|
||||
You should evaluate both the data loader test results and the overall workflow execution. **Approve the code only if both tests pass.**
|
||||
{% endif %}
|
||||
|
||||
## Evaluation Criteria
|
||||
You will be given the standard output (`stdout`) from the data loader test and, if applicable, the workflow test.
|
||||
|
||||
## Exploratory Data Analysis (EDA) Part evaluation
|
||||
- The code has also generated some EDA output to help understand the data better.
|
||||
- The EDA part should be drafted in plain text sending to standard output with command print or other similar functions with no more than ten thousand characters in the following schema:
|
||||
=== Start of EDA part ===
|
||||
{ You EDA output content }
|
||||
=== End of EDA part ===
|
||||
User will use the following code to match: re.search(r"(.*?)=== Start of EDA part ===(.*)=== End of EDA part ===", stdout, re.DOTALL).groups()[1]
|
||||
- The EDA part should include but not limited in the following information in plain text:
|
||||
- The shape of the data.
|
||||
- The first 5 rows of the data.
|
||||
- The data types of each column.
|
||||
- The number of missing values in each column.
|
||||
- The number of unique values in each column.
|
||||
- The distribution of the target variable.
|
||||
- Any other information that you think is important for the following steps.
|
||||
You will be given the EDA output, your job is to check whether the output contains the required and sufficient information. If no EDA output is provided, you should consider it as a failure. Put this evaluation result in the return_checking part.
|
||||
|
||||
Your response must follow this structured JSON format:
|
||||
```json
|
||||
{
|
||||
"execution": "Describe how well the data loader executed, including any errors or issues encountered. Append all error messages and full traceback details without summarizing or omitting any information.",
|
||||
"return_checking": "Evaluate the correctness and integrity of the loaded data. Check for issues like missing values, incorrect data types, outliers, or formatting inconsistencies.",
|
||||
"code": "Assess code quality, readability, and adherence to best practices. Consider efficiency, including whether the code utilizes multi-threading or GPU acceleration for faster data loading.",
|
||||
"final_decision": <true/false>
|
||||
}
|
||||
```
|
||||
|
||||
user: |-
|
||||
--------- Data loader test stdout ---------
|
||||
{{ stdout }}
|
||||
--------- Data loader EDA stdout ---------
|
||||
{% if eda_output is not none %}
|
||||
{{ eda_output }}
|
||||
{% else %}
|
||||
No EDA output is provided.
|
||||
{% endif %}
|
||||
{% if workflow_stdout is not none %}
|
||||
--------- Whole workflow test stdout ---------
|
||||
{{ workflow_stdout }}
|
||||
{% endif %}
|
||||
@@ -0,0 +1,30 @@
|
||||
"""
|
||||
Helper functions for testing the raw_data_loader coder(CoSTEER-based) component.
|
||||
- Does the developer loop work correctly
|
||||
|
||||
It is NOT:
|
||||
- it is not interface unittest(i.e. workspace evaluator in the CoSTEER Loop)
|
||||
"""
|
||||
|
||||
from rdagent.components.coder.data_science.raw_data_loader import DataLoaderCoSTEER
|
||||
from rdagent.components.coder.data_science.raw_data_loader.exp import DataLoaderTask
|
||||
from rdagent.scenarios.data_science.experiment.experiment import DSExperiment
|
||||
from rdagent.scenarios.data_science.scen import KaggleScen
|
||||
|
||||
|
||||
def develop_one_competition(competition: str): # -> experiment
|
||||
scen = KaggleScen(competition=competition)
|
||||
data_loader_coder = DataLoaderCoSTEER(scen)
|
||||
|
||||
# Create the experiment
|
||||
dlt = DataLoaderTask(name="DataLoaderTask", description="")
|
||||
exp = DSExperiment(
|
||||
sub_tasks=[dlt],
|
||||
)
|
||||
|
||||
# Develop the experiment
|
||||
exp = data_loader_coder.develop(exp)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
develop_one_competition("aerial-cactus-identification")
|
||||
@@ -0,0 +1,37 @@
|
||||
"""
|
||||
Developers concentrating on writing documents for a workspace
|
||||
"""
|
||||
|
||||
from rdagent.core.developer import Developer
|
||||
from rdagent.core.experiment import Experiment, FBWorkspace
|
||||
from rdagent.oai.llm_utils import APIBackend
|
||||
from rdagent.utils.agent.ret import MarkdownAgentOut
|
||||
from rdagent.utils.agent.tpl import T
|
||||
|
||||
|
||||
class DocDev(Developer[Experiment]):
|
||||
"""
|
||||
The developer is responsible for writing documents for a workspace.
|
||||
"""
|
||||
|
||||
def develop(self, exp: Experiment) -> None:
|
||||
"""
|
||||
Write documents for the workspace.
|
||||
"""
|
||||
ws: FBWorkspace = exp.experiment_workspace
|
||||
|
||||
file_li = [str(file.relative_to(ws.workspace_path)) for file in ws.workspace_path.rglob("*") if file.is_file()]
|
||||
|
||||
key_file_list = ["main.py", "scores.csv"]
|
||||
|
||||
system_prompt = T(".prompts:docdev.system").r()
|
||||
user_prompt = T(".prompts:docdev.user").r(
|
||||
file_li=file_li,
|
||||
key_files={f: (ws.workspace_path / f).read_text() for f in key_file_list},
|
||||
)
|
||||
|
||||
resp = APIBackend().build_messages_and_create_chat_completion(
|
||||
user_prompt=user_prompt, system_prompt=system_prompt
|
||||
)
|
||||
markdown = MarkdownAgentOut.extract_output(resp)
|
||||
ws.inject_files(**{"README.md": markdown})
|
||||
@@ -0,0 +1,147 @@
|
||||
from pathlib import Path
|
||||
from typing import Literal
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from rdagent.app.data_science.conf import DS_RD_SETTING
|
||||
from rdagent.components.coder.CoSTEER import CoSTEERMultiFeedback
|
||||
from rdagent.components.coder.CoSTEER.evaluators import (
|
||||
CoSTEEREvaluator,
|
||||
CoSTEERSingleFeedback,
|
||||
)
|
||||
from rdagent.components.coder.data_science.conf import get_clear_ws_cmd, get_ds_env
|
||||
from rdagent.components.coder.data_science.utils import remove_eda_part
|
||||
from rdagent.core.experiment import FBWorkspace, Task
|
||||
from rdagent.core.scenario import Scenario
|
||||
from rdagent.utils.agent.tpl import T
|
||||
from rdagent.utils.agent.workflow import build_cls_from_json_with_retry
|
||||
|
||||
DIRNAME = Path(__file__).absolute().resolve().parent
|
||||
|
||||
PipelineSingleFeedback = CoSTEERSingleFeedback
|
||||
PipelineMultiFeedback = CoSTEERMultiFeedback
|
||||
|
||||
NO_SUB = "<No submission.csv file found.>"
|
||||
NO_SCORE = "<No scores.csv file found.>"
|
||||
|
||||
|
||||
class ModelDumpEvaluator(CoSTEEREvaluator):
|
||||
"""This evaluator assumes that it runs after the model"""
|
||||
|
||||
def __init__(self, scen: Scenario, data_type: Literal["sample", "full"]):
|
||||
super().__init__(scen)
|
||||
self.data_type = data_type
|
||||
|
||||
def evaluate(
|
||||
self, target_task: Task, implementation: FBWorkspace, gt_implementation: FBWorkspace, *kargs, **kwargs
|
||||
) -> CoSTEERSingleFeedback:
|
||||
|
||||
model_folder = implementation.workspace_path / "models"
|
||||
# 1) Check if the model_folder is not empty
|
||||
if not model_folder.exists() or not any(model_folder.iterdir()):
|
||||
err_msg = "Model folder (`models` sub folder) is empty or does not exist. The model is not dumped."
|
||||
return CoSTEERSingleFeedback(
|
||||
execution=err_msg,
|
||||
return_checking=err_msg,
|
||||
code=err_msg,
|
||||
final_decision=False,
|
||||
)
|
||||
|
||||
data_source_path = (
|
||||
f"{DS_RD_SETTING.local_data_path}/{self.scen.competition}"
|
||||
if self.data_type == "full"
|
||||
else self.scen.debug_path
|
||||
)
|
||||
env = get_ds_env(
|
||||
extra_volumes={data_source_path: T("scenarios.data_science.share:scen.input_path").r()},
|
||||
running_timeout_period=(
|
||||
DS_RD_SETTING.full_timeout if self.data_type == "full" else DS_RD_SETTING.debug_timeout
|
||||
),
|
||||
)
|
||||
|
||||
# 2) check the result and stdout after reruning the model.
|
||||
|
||||
# Read the content of files submission.csv and scores.csv before execution
|
||||
submission_content_before = (
|
||||
(implementation.workspace_path / "submission.csv").read_text()
|
||||
if (implementation.workspace_path / "submission.csv").exists()
|
||||
else NO_SUB
|
||||
)
|
||||
scores_content_before = (
|
||||
(implementation.workspace_path / "scores.csv").read_text()
|
||||
if (implementation.workspace_path / "scores.csv").exists()
|
||||
else NO_SCORE
|
||||
)
|
||||
|
||||
# Remove the files submission.csv and scores.csv
|
||||
implementation.execute(env=env, entry=get_clear_ws_cmd(stage="before_inference"))
|
||||
|
||||
# Execute the main script
|
||||
stdout = remove_eda_part(implementation.execute(env=env, entry="python main.py"))
|
||||
|
||||
# walk model_folder and list the files
|
||||
model_folder_files = [
|
||||
str(file.relative_to(implementation.workspace_path)) for file in model_folder.iterdir() if file.is_file()
|
||||
]
|
||||
|
||||
# this will assert the generation of necessary files
|
||||
for f in ["submission.csv", "scores.csv"]:
|
||||
if not (implementation.workspace_path / f).exists():
|
||||
err_msg = f"{f} does not exist. The model is not dumped. Make sure that the required files, like submission.csv and scores.csv, are created even if you bypass the model training step by loading the saved model file directly."
|
||||
return CoSTEERSingleFeedback(
|
||||
execution=err_msg,
|
||||
return_checking=err_msg,
|
||||
code=err_msg,
|
||||
final_decision=False,
|
||||
)
|
||||
|
||||
# Check if scores contain NaN (values)
|
||||
score_df = pd.read_csv((implementation.workspace_path / "scores.csv"), index_col=0)
|
||||
if score_df.isnull().values.any():
|
||||
nan_locations = score_df[score_df.isnull().any(axis=1)]
|
||||
err_msg = f"\n[Error] The scores dataframe contains NaN values at the following locations:\n{nan_locations}"
|
||||
return CoSTEERSingleFeedback(
|
||||
execution=err_msg,
|
||||
return_checking=err_msg,
|
||||
code=err_msg,
|
||||
final_decision=False,
|
||||
)
|
||||
|
||||
submission_content_after = (
|
||||
(implementation.workspace_path / "submission.csv").read_text()
|
||||
if (implementation.workspace_path / "submission.csv").exists()
|
||||
else NO_SUB
|
||||
)
|
||||
scores_content_after = (
|
||||
(implementation.workspace_path / "scores.csv").read_text()
|
||||
if (implementation.workspace_path / "scores.csv").exists()
|
||||
else NO_SCORE
|
||||
)
|
||||
|
||||
system_prompt = T(".prompts:dump_model_eval.system").r()
|
||||
user_prompt = T(".prompts:dump_model_eval.user").r(
|
||||
stdout=stdout.strip(),
|
||||
code=implementation.all_codes,
|
||||
model_folder_files=model_folder_files,
|
||||
scores_content_before=scores_content_before,
|
||||
scores_content_after=scores_content_after,
|
||||
)
|
||||
|
||||
csfb = build_cls_from_json_with_retry(
|
||||
CoSTEERSingleFeedback,
|
||||
system_prompt=system_prompt,
|
||||
user_prompt=user_prompt,
|
||||
)
|
||||
|
||||
if DS_RD_SETTING.model_dump_check_level == "high":
|
||||
# Read the content of files submission.csv and scores.csv after execution
|
||||
# Check if the content has changed
|
||||
# excactly same checking. But it will take more user's time
|
||||
if scores_content_before != scores_content_after:
|
||||
return_msg = "\n[Error] The content of scores.csv has changed. Please check the code to ensure that the model is dumped correctly, and rerun the code to use the model directly without retraining it."
|
||||
return_msg += f"\nBefore:\n{scores_content_before}\nAfter:\n{scores_content_after}"
|
||||
if submission_content_before != submission_content_after:
|
||||
# If the scores file changes, display the two contents and append it into the return_checking
|
||||
return_msg = "[Error] The content of submission.csv has changed. Please check the code to ensure that the model is dumped correctly, and rerun the code to use the model directly without retraining it."
|
||||
csfb.return_checking = (csfb.return_checking or "") + return_msg
|
||||
return csfb
|
||||
@@ -0,0 +1,91 @@
|
||||
dump_model_coder:
|
||||
guideline: |-
|
||||
Please dump the model in a "models/" subfolder in the first running, and the script rerun performs inference without needing to retrain the model when running the code again.
|
||||
If there are parameters generated from the training data that might be needed for inference on test data, please save them in the "models/" subfolder as well.
|
||||
If no test set is provided, reserve a portion of the data as your test set and save the generated test files in the models/ subfolder for use in submission and inference.
|
||||
Make sure that the required files, like submission.csv and scores.csv, are created without model training step through loading the saved model and test data file directly.
|
||||
|
||||
dump_model_eval:
|
||||
system: |-
|
||||
You are a data scientist tasked with evaluating code generation. You've developed a Kaggle competition code that can produce a submission file.
|
||||
The code should follow the guideline below:
|
||||
{% include "components.coder.data_science.share.prompts:dump_model_coder.guideline" %}
|
||||
|
||||
You will receive the following information:
|
||||
- The implemented code
|
||||
- The stdout from running the code
|
||||
- The file list in "models/" subfolder
|
||||
- The scores.csv file generated during both training and inference (if it exists)
|
||||
|
||||
Focus on these aspects:
|
||||
- Check if the code saves the model in the "models/" subfolder.
|
||||
- Check if the code saves the test data in the "models/" subfolder when there is no test data specified.
|
||||
- Ensure that when the code is rerun, it skips the training process and loads the model from the "models/" subfolder for direct inference.
|
||||
- Verify that there is no training activity in the output.
|
||||
- Ensure that even if you skip the model training by loading saved models, the files like scores.csv and submission.csv are still correctly created.
|
||||
- The model's performance should remain consistent and not vary unreasonably between training and inference.
|
||||
|
||||
Please respond with your feedback in the following JSON format and order
|
||||
```json
|
||||
{
|
||||
"execution": "Describe whether the code executed successfully. Include any errors or issues encountered, and append all error messages and full traceback details without summarizing or omitting any information. Carefully check the stdout to ensure that when the code is rerun, it skips the training process and loads the model from the 'models/' subfolder for direct inference. Append the information that makes you think that the model is still being retrained when rerunning the code."
|
||||
"return_checking": "Verify the generated files include necessary files. Make sure scores.csv file does not change unreasonably between training and inference",
|
||||
"code": "The code has explicity dump the model into 'models/' subfolder; When the modes files are already in 'models/' subfolder, the code will explicity skip the training process.",
|
||||
"final_decision": <true or false in boolean type; only return true when ensuring that the code saves the model in a 'models/' subfolder, and the script rerun performs inference without needing to retrain the model.>
|
||||
}
|
||||
```
|
||||
|
||||
user: |-
|
||||
------------ The implemented code ------------
|
||||
{{code}}
|
||||
|
||||
------------ The stdout from running the code ------------
|
||||
{{stdout}}
|
||||
|
||||
------------ The file list in "models/" subfolder ------------
|
||||
{% for f in model_folder_files %}
|
||||
- {{ f }}
|
||||
{% endfor %}
|
||||
|
||||
------------ The scores.csv file generated ------------
|
||||
# Training:
|
||||
{{scores_content_before}}
|
||||
|
||||
# Inference:
|
||||
{{scores_content_after}}
|
||||
|
||||
|
||||
docdev:
|
||||
system: |-
|
||||
{% include "scenarios.data_science.share:scen.role" %} Your task is to create documentation for a data science solution.
|
||||
|
||||
You will be given:
|
||||
- a list of files in the folder.
|
||||
- content from some important files.
|
||||
|
||||
Please explain the trained models in the "models/" folder. The training and inference processes are detailed in the `main.py` file. The models' evaluation results are in `scores.csv`. Please respond with a markdown file that includes the following information:
|
||||
- Explain the purpose of each model. If some models are part of a group (like those from cross-validation), describe them together.
|
||||
- Provide key details for each model group:
|
||||
- Important training parameters
|
||||
- Model details
|
||||
- Performance of each model
|
||||
|
||||
Be brief. Mention the file path when you introduce files.
|
||||
Don't introduce anything other than models.
|
||||
|
||||
{% include "utils.agent.tpl:MarkdownOut" %}
|
||||
|
||||
user: |-
|
||||
--------------- The file list in the workspace ---------------
|
||||
{% for f in file_li %}
|
||||
- {{ f }}
|
||||
{% endfor %}
|
||||
|
||||
--------------- File content of each file ---------------
|
||||
{% for fname, content in key_files.items() %}
|
||||
File Path: {{fname}}
|
||||
```
|
||||
{{content}}
|
||||
```
|
||||
{% endfor %}
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
import re
|
||||
|
||||
|
||||
def remove_eda_part(stdout: str) -> str:
|
||||
"""Data Science scenario have a LLM-based EDA feature. We can remove it when current task does not involve EDA"""
|
||||
return re.sub(r"=== Start of EDA part ===(.*)=== End of EDA part ===", "", stdout, flags=re.DOTALL)
|
||||
@@ -0,0 +1,135 @@
|
||||
import json
|
||||
from typing import Dict
|
||||
|
||||
from rdagent.app.data_science.conf import DS_RD_SETTING
|
||||
from rdagent.components.coder.CoSTEER import CoSTEER
|
||||
from rdagent.components.coder.CoSTEER.evaluators import (
|
||||
CoSTEERMultiEvaluator,
|
||||
CoSTEERSingleFeedback,
|
||||
)
|
||||
from rdagent.components.coder.CoSTEER.evolving_strategy import (
|
||||
MultiProcessEvolvingStrategy,
|
||||
)
|
||||
from rdagent.components.coder.CoSTEER.knowledge_management import (
|
||||
CoSTEERQueriedKnowledge,
|
||||
)
|
||||
from rdagent.components.coder.data_science.conf import DSCoderCoSTEERSettings
|
||||
from rdagent.components.coder.data_science.workflow.eval import (
|
||||
WorkflowGeneralCaseSpecEvaluator,
|
||||
)
|
||||
from rdagent.components.coder.data_science.workflow.exp import WorkflowTask
|
||||
from rdagent.core.exception import CoderError
|
||||
from rdagent.core.experiment import FBWorkspace
|
||||
from rdagent.core.scenario import Scenario
|
||||
from rdagent.oai.llm_utils import APIBackend
|
||||
from rdagent.utils.agent.ret import PythonAgentOut
|
||||
from rdagent.utils.agent.tpl import T
|
||||
|
||||
|
||||
class WorkflowMultiProcessEvolvingStrategy(MultiProcessEvolvingStrategy):
|
||||
def implement_one_task(
|
||||
self,
|
||||
target_task: WorkflowTask,
|
||||
queried_knowledge: CoSTEERQueriedKnowledge | None = None,
|
||||
workspace: FBWorkspace | None = None,
|
||||
prev_task_feedback: CoSTEERSingleFeedback | None = None,
|
||||
) -> dict[str, str]:
|
||||
workflow_information_str = target_task.get_task_information()
|
||||
|
||||
# 1. query
|
||||
queried_similar_successful_knowledge = (
|
||||
queried_knowledge.task_to_similar_task_successful_knowledge[workflow_information_str]
|
||||
if queried_knowledge is not None
|
||||
else []
|
||||
)
|
||||
queried_former_failed_knowledge = (
|
||||
queried_knowledge.task_to_former_failed_traces[workflow_information_str]
|
||||
if queried_knowledge is not None
|
||||
else []
|
||||
)
|
||||
queried_former_failed_knowledge = (
|
||||
[
|
||||
knowledge
|
||||
for knowledge in queried_former_failed_knowledge[0]
|
||||
if knowledge.implementation.file_dict.get("main.py") != workspace.file_dict.get("main.py")
|
||||
],
|
||||
queried_former_failed_knowledge[1],
|
||||
)
|
||||
|
||||
# 2. code
|
||||
system_prompt = T(".prompts:workflow_coder.system").r(
|
||||
task_desc=workflow_information_str,
|
||||
competition_info=self.scen.get_scenario_all_desc(eda_output=workspace.file_dict.get("EDA.md", None)),
|
||||
queried_similar_successful_knowledge=queried_similar_successful_knowledge,
|
||||
queried_former_failed_knowledge=queried_former_failed_knowledge[0],
|
||||
out_spec=PythonAgentOut.get_spec(),
|
||||
)
|
||||
user_prompt = T(".prompts:workflow_coder.user").r(
|
||||
load_data_code=workspace.file_dict["load_data.py"],
|
||||
feature_code=workspace.file_dict["feature.py"],
|
||||
model_codes=workspace.get_codes(r"^model_(?!test)\w+\.py$"),
|
||||
ensemble_code=workspace.file_dict["ensemble.py"],
|
||||
latest_code=workspace.file_dict.get("main.py"),
|
||||
code_spec=(
|
||||
workspace.file_dict["spec/workflow.md"]
|
||||
if DS_RD_SETTING.spec_enabled
|
||||
else T("scenarios.data_science.share:component_spec.Workflow").r()
|
||||
),
|
||||
latest_code_feedback=prev_task_feedback,
|
||||
)
|
||||
|
||||
for _ in range(5):
|
||||
workflow_code = PythonAgentOut.extract_output(
|
||||
APIBackend().build_messages_and_create_chat_completion(
|
||||
user_prompt=user_prompt,
|
||||
system_prompt=system_prompt,
|
||||
)
|
||||
)
|
||||
if workflow_code != workspace.file_dict.get("main.py"):
|
||||
break
|
||||
else:
|
||||
user_prompt = user_prompt + "\nPlease avoid generating same code to former code!"
|
||||
else:
|
||||
raise CoderError("Failed to generate a new workflow code.")
|
||||
|
||||
return {"main.py": workflow_code}
|
||||
|
||||
def assign_code_list_to_evo(self, code_list: list[dict[str, str]], evo):
|
||||
"""
|
||||
Assign the code list to the evolving item.
|
||||
|
||||
The code list is aligned with the evolving item's sub-tasks.
|
||||
If a task is not implemented, put a None in the list.
|
||||
"""
|
||||
for index in range(len(evo.sub_tasks)):
|
||||
if code_list[index] is None:
|
||||
continue
|
||||
if evo.sub_workspace_list[index] is None:
|
||||
# evo.sub_workspace_list[index] = FBWorkspace(target_task=evo.sub_tasks[index])
|
||||
evo.sub_workspace_list[index] = evo.experiment_workspace
|
||||
evo.sub_workspace_list[index].inject_files(**code_list[index])
|
||||
return evo
|
||||
|
||||
|
||||
class WorkflowCoSTEER(CoSTEER):
|
||||
def __init__(
|
||||
self,
|
||||
scen: Scenario,
|
||||
*args,
|
||||
**kwargs,
|
||||
) -> None:
|
||||
settings = DSCoderCoSTEERSettings()
|
||||
eva = CoSTEERMultiEvaluator(
|
||||
WorkflowGeneralCaseSpecEvaluator(scen=scen), scen=scen
|
||||
) # Please specify whether you agree running your eva in parallel or not
|
||||
es = WorkflowMultiProcessEvolvingStrategy(scen=scen, settings=settings)
|
||||
super().__init__(
|
||||
*args,
|
||||
settings=settings,
|
||||
eva=eva,
|
||||
es=es,
|
||||
evolving_version=2,
|
||||
scen=scen,
|
||||
max_loop=DS_RD_SETTING.coder_max_loop,
|
||||
**kwargs,
|
||||
)
|
||||
@@ -0,0 +1,155 @@
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from rdagent.app.data_science.conf import DS_RD_SETTING
|
||||
from rdagent.components.coder.CoSTEER.evaluators import (
|
||||
CoSTEEREvaluator,
|
||||
CoSTEERMultiFeedback,
|
||||
CoSTEERSingleFeedback,
|
||||
)
|
||||
from rdagent.components.coder.data_science.conf import get_clear_ws_cmd, get_ds_env
|
||||
from rdagent.components.coder.data_science.utils import remove_eda_part
|
||||
from rdagent.core.evolving_framework import QueriedKnowledge
|
||||
from rdagent.core.experiment import FBWorkspace, Task
|
||||
from rdagent.log import rdagent_logger as logger
|
||||
from rdagent.utils.agent.tpl import T
|
||||
from rdagent.utils.agent.workflow import build_cls_from_json_with_retry
|
||||
|
||||
DIRNAME = Path(__file__).absolute().resolve().parent
|
||||
|
||||
WorkflowSingleFeedback = CoSTEERSingleFeedback
|
||||
WorkflowMultiFeedback = CoSTEERMultiFeedback
|
||||
|
||||
|
||||
class WorkflowGeneralCaseSpecEvaluator(CoSTEEREvaluator):
|
||||
"""
|
||||
Motivation case:
|
||||
- Simplest case, we already split the data into train_data, valid_data, and test_data. We require the model to learn (optionally validate on valid data), and infer on test data.
|
||||
|
||||
Test workflow:
|
||||
- Build train, valid, and test data to run it, and test the output (e.g., shape, etc.)
|
||||
"""
|
||||
|
||||
def evaluate(
|
||||
self,
|
||||
target_task: Task,
|
||||
implementation: FBWorkspace,
|
||||
gt_implementation: FBWorkspace,
|
||||
queried_knowledge: QueriedKnowledge = None,
|
||||
**kwargs,
|
||||
) -> CoSTEERSingleFeedback:
|
||||
target_task_information = target_task.get_task_information()
|
||||
if (
|
||||
queried_knowledge is not None
|
||||
and target_task_information in queried_knowledge.success_task_to_knowledge_dict
|
||||
):
|
||||
return queried_knowledge.success_task_to_knowledge_dict[target_task_information].feedback
|
||||
elif queried_knowledge is not None and target_task_information in queried_knowledge.failed_task_info_set:
|
||||
return WorkflowSingleFeedback(
|
||||
execution="This task has failed too many times, skip implementation.",
|
||||
return_checking="This task has failed too many times, skip implementation.",
|
||||
code="This task has failed too many times, skip implementation.",
|
||||
final_decision=False,
|
||||
)
|
||||
|
||||
env = get_ds_env(extra_volumes={self.scen.debug_path: T("scenarios.data_science.share:scen.input_path").r()})
|
||||
|
||||
# # DockerEnv for MLEBench submission validation
|
||||
# mle_de_conf = MLEBDockerConf()
|
||||
# mle_de_conf.extra_volumes = {
|
||||
# f"{DS_RD_SETTING.local_data_path}/zip_files": "/mle/data",
|
||||
# }
|
||||
# mde = DockerEnv(conf=mle_de_conf)
|
||||
# mde.prepare()
|
||||
|
||||
# Clean the scores.csv & submission.csv.
|
||||
implementation.execute(env=env, entry=get_clear_ws_cmd())
|
||||
|
||||
stdout = implementation.execute(env=env, entry=f"python -m coverage run main.py")
|
||||
|
||||
# remove EDA part
|
||||
stdout = remove_eda_part(stdout)
|
||||
|
||||
# Check score file
|
||||
score_fp = implementation.workspace_path / "scores.csv"
|
||||
score_ret_code = 0
|
||||
score_check_text = ""
|
||||
if not score_fp.exists():
|
||||
score_check_text = "[Error] Metrics file (scores.csv) is not generated!"
|
||||
score_ret_code = 1
|
||||
implementation.execute(env=env, entry="python -m coverage json -o coverage.json")
|
||||
coverage_report_path = implementation.workspace_path / "coverage.json"
|
||||
if coverage_report_path.exists():
|
||||
used_files = set(json.loads(coverage_report_path.read_text())["files"].keys())
|
||||
coverage_report_path.unlink()
|
||||
logger.info(f"All used scripts: {used_files}")
|
||||
if len(used_files) == 1:
|
||||
score_check_text += f"\n[Error] The only used script is {used_files}.\nPlease check if you have implemented entry point in 'main.py'."
|
||||
else:
|
||||
try:
|
||||
score_df = pd.read_csv(score_fp, index_col=0)
|
||||
model_set_in_scores = set(score_df.index)
|
||||
# We assume that model names in `score_df` are stored without the '.py' file extension.
|
||||
model_set_in_folder = set(
|
||||
f[:-3] for f in implementation.file_dict.keys() if re.match(r"^model_(?!test)\w+\.py$", f)
|
||||
)
|
||||
|
||||
# Check model names (index)
|
||||
if model_set_in_scores != model_set_in_folder.union({"ensemble"}):
|
||||
score_check_text += f"\n[Error] The scores dataframe does not contain the correct model names as index.\ncorrect model names are: {model_set_in_folder.union({'ensemble'})}\nscore_df is:\n{score_df}"
|
||||
score_ret_code = 1
|
||||
|
||||
# Check metric name (columns)
|
||||
if score_df.columns.tolist() != [self.scen.metric_name]:
|
||||
score_check_text += f"\n[Error] The scores dataframe does not contain the correct column names.\nCorrect columns is: ['{self.scen.metric_name}']\nBut got: {score_df.columns.tolist()}"
|
||||
score_ret_code = 1
|
||||
|
||||
# Check if scores contain NaN (values)
|
||||
if score_df.isnull().values.any():
|
||||
nan_locations = score_df[score_df.isnull().any(axis=1)]
|
||||
score_check_text += f"\n[Error] The scores dataframe contains NaN values at the following locations:\n{nan_locations}"
|
||||
score_ret_code = 1
|
||||
|
||||
except Exception as e:
|
||||
score_check_text += f"\n[Error] in checking the scores.csv file: {e}\nscores.csv's content:\n-----\n{score_fp.read_text()}\n-----"
|
||||
score_ret_code = 1
|
||||
|
||||
# Check submission file
|
||||
base_check_code = T(".eval_tests.submission_format_test", ftype="txt").r()
|
||||
implementation.inject_files(**{"test/submission_format_test.py": base_check_code})
|
||||
# stdout += "----Submission Check 1-----\n"
|
||||
submission_check_out, submission_ret_code = implementation.execute_ret_code(
|
||||
env=env, entry="python test/submission_format_test.py"
|
||||
)
|
||||
stdout += "\n" + submission_check_out
|
||||
|
||||
system_prompt = T(".prompts:workflow_eval.system").r(
|
||||
# here we pass `None` to `eda_output` because we do not have nor need EDA output for workflow.
|
||||
scenario=self.scen.get_scenario_all_desc(eda_output=None),
|
||||
task_desc=target_task.get_task_information(),
|
||||
spec=(
|
||||
implementation.file_dict["spec/workflow.md"]
|
||||
if DS_RD_SETTING.spec_enabled
|
||||
else T("scenarios.data_science.share:component_spec.Workflow").r()
|
||||
),
|
||||
)
|
||||
user_prompt = T(".prompts:workflow_eval.user").r(
|
||||
stdout=stdout.strip(),
|
||||
code=implementation.file_dict["main.py"],
|
||||
)
|
||||
wfb = build_cls_from_json_with_retry(
|
||||
WorkflowSingleFeedback,
|
||||
system_prompt=system_prompt,
|
||||
user_prompt=user_prompt,
|
||||
init_kwargs_update_func=WorkflowSingleFeedback.val_and_update_init_dict,
|
||||
)
|
||||
if score_ret_code != 0:
|
||||
wfb.final_decision = False
|
||||
wfb.return_checking += "\n" + score_check_text
|
||||
if submission_ret_code != 0:
|
||||
wfb.final_decision = False
|
||||
wfb.return_checking += "\nSubmission file check failed."
|
||||
return wfb
|
||||
@@ -0,0 +1,77 @@
|
||||
from pathlib import Path
|
||||
import pandas as pd
|
||||
import hashlib
|
||||
|
||||
def calculate_md5(file_path):
|
||||
with open(file_path, "rb") as f:
|
||||
file_hash = hashlib.md5(f.read()).hexdigest()
|
||||
return file_hash
|
||||
|
||||
file_md5 = calculate_md5("scores.csv")
|
||||
|
||||
"""
|
||||
find . | grep -i sample | grep -i submission | grep -v sample_submission.csv | grep -v zip_files | grep -v 'sample/'
|
||||
./denoising-dirty-documents/sampleSubmission.csv
|
||||
./the-icml-2013-whale-challenge-right-whale-redux/sampleSubmission.csv
|
||||
./text-normalization-challenge-russian-language/ru_sample_submission_2.csv.zip
|
||||
./text-normalization-challenge-russian-language/ru_sample_submission_2.csv
|
||||
./random-acts-of-pizza/sampleSubmission.csv
|
||||
./text-normalization-challenge-english-language/en_sample_submission_2.csv.zip
|
||||
./text-normalization-challenge-english-language/en_sample_submission_2.csv
|
||||
./detecting-insults-in-social-commentary/sample_submission_null.csv
|
||||
"""
|
||||
|
||||
# Find sample submission file dynamically
|
||||
input_dir = Path("{% include "scenarios.data_science.share:scen.input_path" %}")
|
||||
# Look for common variations of sample submission filenames
|
||||
sample_submission_files = list(input_dir.glob("*sample_submission*.csv")) + \
|
||||
list(input_dir.glob("*sampleSubmission*.csv"))
|
||||
|
||||
assert sample_submission_files, "Error: No sample submission file found in {% include "scenarios.data_science.share:scen.input_path" %}"
|
||||
|
||||
# Use first matching file
|
||||
sample_submission_name = sample_submission_files[0].name
|
||||
SAMPLE_SUBMISSION_PATH = str(sample_submission_files[0])
|
||||
print(f"Using sample submission file: {sample_submission_name}")
|
||||
|
||||
# Check if the sample submission file exists
|
||||
assert Path(SAMPLE_SUBMISSION_PATH).exists(), f"Error: {sample_submission_name} not found at {SAMPLE_SUBMISSION_PATH}"
|
||||
|
||||
# Check if our submission file exists
|
||||
assert Path('submission.csv').exists(), "Error: submission.csv not found"
|
||||
|
||||
sample_submission = pd.read_csv(SAMPLE_SUBMISSION_PATH)
|
||||
our_submission = pd.read_csv('submission.csv')
|
||||
|
||||
success = True
|
||||
# Print the columns of the sample submission file
|
||||
print(f"Columns in {sample_submission_name}:", sample_submission.columns)
|
||||
print("Columns in our_submission.csv:", our_submission.columns)
|
||||
|
||||
for col in sample_submission.columns:
|
||||
if col not in our_submission.columns:
|
||||
success = False
|
||||
print(f'Column {col} not found in submission.csv')
|
||||
|
||||
if success:
|
||||
print(f'submission.csv\'s columns aligns with {sample_submission_name} .')
|
||||
|
||||
|
||||
# Print the first 5 rows of the two submission files, with columns separated by commas.
|
||||
def print_first_rows(file_path, file_name, num_rows=5):
|
||||
print(f"\nFirst {num_rows} rows of {file_name}:")
|
||||
try:
|
||||
with open(file_path, 'r') as file:
|
||||
for i, line in enumerate(file):
|
||||
if i < num_rows:
|
||||
print(line.strip())
|
||||
else:
|
||||
break
|
||||
except FileNotFoundError:
|
||||
print(f"Error: {file_name} not found.")
|
||||
|
||||
print_first_rows(SAMPLE_SUBMISSION_PATH, sample_submission_name)
|
||||
print_first_rows('submission.csv', 'submission.csv')
|
||||
|
||||
assert calculate_md5("scores.csv") == file_md5, "scores.csv should not be rewritten"
|
||||
print(f"\nPlease Checked the content of the submission file(submission.csv should align with {sample_submission_name}). ")
|
||||
@@ -0,0 +1,14 @@
|
||||
import pickle
|
||||
import site
|
||||
import traceback
|
||||
from pathlib import Path
|
||||
from typing import Dict, Optional
|
||||
|
||||
from rdagent.components.coder.CoSTEER.task import CoSTEERTask
|
||||
from rdagent.core.utils import cache_with_pickle
|
||||
|
||||
|
||||
# Because we use isinstance to distinguish between different types of tasks, we need to use sub classes to represent different types of tasks
|
||||
class WorkflowTask(CoSTEERTask):
|
||||
def __init__(self, name: str = "Workflow", *args, **kwargs) -> None:
|
||||
super().__init__(name=name, *args, **kwargs)
|
||||
@@ -0,0 +1,137 @@
|
||||
workflow_coder:
|
||||
system: |-
|
||||
You are a world-class data scientist and machine learning engineer with deep expertise in statistics, mathematics, and computer science.
|
||||
Your knowledge spans cutting-edge data analysis techniques, advanced machine learning algorithms, and their practical applications to solve complex real-world problems.
|
||||
|
||||
## Task Description
|
||||
{{ task_desc }}
|
||||
|
||||
Here is the competition information for this task:
|
||||
{{ competition_info }}
|
||||
|
||||
{% if queried_similar_successful_knowledge|length != 0 or queried_former_failed_knowledge|length != 0 %}
|
||||
## Relevant Information for This Task
|
||||
{% endif %}
|
||||
|
||||
{% if queried_similar_successful_knowledge|length != 0 %}
|
||||
--------- Successful Implementations for Similar Models ---------
|
||||
====={% for similar_successful_knowledge in queried_similar_successful_knowledge %} Model {{ loop.index }}:=====
|
||||
{{ similar_successful_knowledge.target_task.get_task_information() }}
|
||||
=====Code:=====
|
||||
{{ similar_successful_knowledge.implementation.file_dict["main.py"] }}
|
||||
{% endfor %}
|
||||
{% endif %}
|
||||
|
||||
{% if queried_former_failed_knowledge|length != 0 %}
|
||||
--------- Previous Failed Attempts ---------
|
||||
{% for former_failed_knowledge in queried_former_failed_knowledge %} Attempt {{ loop.index }}:
|
||||
=====Code:=====
|
||||
{{ former_failed_knowledge.implementation.file_dict["main.py"] }}
|
||||
=====Feedback:=====
|
||||
{{ former_failed_knowledge.feedback }}
|
||||
{% endfor %}
|
||||
{% endif %}
|
||||
|
||||
## Guidelines
|
||||
1. Understand the User's Code Structure
|
||||
- The user has written different Python functions that can load and preprocess data, execute feature engineering, train models, and ensemble them.
|
||||
- Each functionality is in a separate Python file.
|
||||
2. Your task is only to integrate the existing processes of load_data, feature, model, and ensemble into a complete workflow. Do not edit or modify the existing Python files. The final step should output the predictions in the required format.
|
||||
3. The user may provide specific code organization rules and instructions. Ensure that the integration follows the given framework and structure.
|
||||
4. After predicting the output, print the shape and other information of the output to stdout to help the evaluator assess the code.
|
||||
5. You should avoid using logging module to output information in your generated code, and instead use the print() function.
|
||||
{% include "scenarios.data_science.share:guidelines.coding" %}
|
||||
|
||||
## Output Format
|
||||
{% if out_spec %}
|
||||
{{ out_spec }}
|
||||
{% else %}
|
||||
Please response the code in the following json format. Here is an example structure for the JSON output:
|
||||
{
|
||||
"code": "The Python code as a string."
|
||||
}
|
||||
{% endif %}
|
||||
|
||||
user: |-
|
||||
--------- Code Specification ---------
|
||||
{{ code_spec }}
|
||||
|
||||
--------- load data code ---------
|
||||
file: load_data.py
|
||||
{{ load_data_code }}
|
||||
|
||||
--------- feature engineering code ---------
|
||||
file: feature.py
|
||||
{{ feature_code }}
|
||||
|
||||
--------- model training code ---------
|
||||
Attention: The input and output of the model function is flexible. Training dataset is necessary, but validation and test dateset might be optional. The hyperparameters can either be passed as arguments or be set as default values in the function. You need to use the function correctly.
|
||||
All model files share the same function name. Please import the model files with their name like: from {file_name} import {function_name}
|
||||
{{ model_codes }}
|
||||
|
||||
--------- ensemble code ---------
|
||||
Note, we will check the index of the score.csv, so please use the model name as the index to feed into ensemble function.
|
||||
file: ensemble.py
|
||||
{{ ensemble_code }}
|
||||
|
||||
{% if latest_code %}
|
||||
--------- Former code ---------
|
||||
{{ latest_code }}
|
||||
{% if latest_code_feedback is not none %}
|
||||
--------- Feedback to former code ---------
|
||||
{{ latest_code_feedback }}
|
||||
{% endif %}
|
||||
The former code contains errors. You should correct the code based on the provided information, ensuring you do not repeat the same mistakes.
|
||||
{% endif %}
|
||||
|
||||
workflow_eval:
|
||||
system: |-
|
||||
You are a data scientist responsible for evaluating workflow code generation.
|
||||
|
||||
## Task Description
|
||||
The user is trying to build a workflow in the following scenario:
|
||||
{{ scenario }}
|
||||
|
||||
The main code generation task is as follows:
|
||||
{{ task_desc }}
|
||||
|
||||
The user provides workflow information and its components.
|
||||
The details on how to structure the workflow are given in the specification file:
|
||||
```markdown
|
||||
{{ spec }}
|
||||
```
|
||||
|
||||
This workflow integrates multiple stages, including:
|
||||
- Data loading
|
||||
- Feature engineering
|
||||
- Model training
|
||||
- Ensembling
|
||||
|
||||
## Evaluation Scope
|
||||
Your focus is to check whether the workflow code:
|
||||
1. Executes successfully, correctly organizing components and generating a final submission.
|
||||
2. Generates predictions in the correct format, ensuring they align with the **sample submission** structure!
|
||||
|
||||
[Note]
|
||||
1. The individual components (data loading, feature engineering, model tuning, etc.) have already been evaluated by the user. You should only evaluate and improve the workflow code, unless there are critical issues in the components.
|
||||
2. Model performance is NOT a concern in this evaluation—only correct execution and formatting matter.
|
||||
3. As long as the execution does not exceed the time limit, ensure that the code uses cross-validation to split the training data and train the model. If cross-validation is not used, mention it in the execution section and set `final_decision` to `false`.
|
||||
|
||||
## Evaluation Criteria
|
||||
You will be given the workflow execution output (`stdout`) to determine correctness.
|
||||
|
||||
Please respond with your feedback in the following JSON format and order
|
||||
```json
|
||||
{
|
||||
"execution": "Describe whether the main workflow executed successfully, correctly integrating all components and generating the final submission. Include any errors or issues encountered, and append all error messages and full traceback details without summarizing or omitting any information.",
|
||||
"return_checking": "Verify the generated files, particularly the submission file. Ensure that its format matches the sample submission, checking the index, column names, and CSV content.",
|
||||
"code": "Provide feedback on code quality, readability, and adherence to the given specifications.",
|
||||
"final_decision": <true/false>
|
||||
}
|
||||
```
|
||||
|
||||
user: |-
|
||||
--------- Workflow test stdout ---------
|
||||
{{ stdout }}
|
||||
--------- Workflow code generated by user ---------
|
||||
{{ code }}
|
||||
@@ -0,0 +1,59 @@
|
||||
"""
|
||||
Generate dataset to test the workflow output
|
||||
"""
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from rdagent.components.coder.CoSTEER.config import CoSTEER_SETTINGS
|
||||
from rdagent.components.coder.data_science.workflow import WorkflowCoSTEER
|
||||
from rdagent.components.coder.data_science.workflow.eval import (
|
||||
WorkflowGeneralCaseSpecEvaluator,
|
||||
)
|
||||
from rdagent.components.coder.data_science.workflow.exp import WorkflowTask
|
||||
from rdagent.core.experiment import FBWorkspace
|
||||
from rdagent.scenarios.data_science.experiment.experiment import DSExperiment
|
||||
from rdagent.scenarios.data_science.scen import KaggleScen
|
||||
|
||||
|
||||
def develop_one_competition(competition: str):
|
||||
scen = KaggleScen(competition=competition)
|
||||
workflow_coder = WorkflowCoSTEER(scen)
|
||||
|
||||
wt = WorkflowTask(
|
||||
name="WorkflowTask",
|
||||
description="Integrate the existing processes of load_data, feature, model, and ensemble into a complete workflow.",
|
||||
base_code="",
|
||||
)
|
||||
|
||||
tpl_ex_path = Path(__file__).resolve() / Path("rdagent/scenarios/kaggle/tpl_ex").resolve() / competition
|
||||
injected_file_names = ["spec/workflow.md", "load_data.py", "feature.py", "model01.py", "ensemble.py", "main.py"]
|
||||
|
||||
workflowexp = FBWorkspace()
|
||||
for file_name in injected_file_names:
|
||||
file_path = tpl_ex_path / file_name
|
||||
workflowexp.inject_files(**{file_name: file_path.read_text()})
|
||||
|
||||
wt.base_code += workflowexp.file_dict["main.py"]
|
||||
exp = DSExperiment(
|
||||
sub_tasks=[wt],
|
||||
)
|
||||
|
||||
"""es = WorkflowMultiProcessEvolvingStrategy(scen=scen, settings=CoSTEER_SETTINGS)
|
||||
new_code = es.implement_one_task(target_task=wt, queried_knowledge=None, workspace = workflowexp)
|
||||
print(new_code)"""
|
||||
|
||||
"""eva = WorkflowGeneralCaseSpecEvaluator(scen=scen)
|
||||
exp.feedback = eva.evaluate(target_task=wt, queried_knowledge=None, implementation=workflowexp, gt_implementation=None)
|
||||
print(exp.feedback)"""
|
||||
|
||||
# Run the experiment
|
||||
for file_name in injected_file_names:
|
||||
file_path = tpl_ex_path / file_name
|
||||
exp.experiment_workspace.inject_files(**{file_name: file_path.read_text()})
|
||||
|
||||
exp = workflow_coder.develop(exp)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
develop_one_competition("aerial-cactus-identification")
|
||||
# dotenv run -- python rdagent/components/coder/data_science/workflow/test.py
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user