chore: ui change (#844)

* ui change

* ui change
This commit is contained in:
XianBW
2025-05-06 11:34:21 +08:00
committed by GitHub
parent 6b7988fba1
commit aa97d5a09d
3 changed files with 173 additions and 59 deletions
+81 -7
View File
@@ -11,6 +11,7 @@ import plotly.graph_objects as go
import streamlit as st
from streamlit import session_state as state
from rdagent.log.mle_summary import extract_mle_json
from rdagent.log.ui.conf import UI_SETTING
from rdagent.log.ui.ds_trace import load_times
from rdagent.scenarios.kaggle.kaggle_crawler import leaderboard_scores
@@ -38,6 +39,7 @@ def get_script_time(stdout_p: Path):
return None
@st.cache_data(persist=True)
def get_final_sota_exp(log_path: Path):
sota_exp_paths = [i for i in log_path.rglob(f"**/SOTA experiment/**/*.pkl")]
if len(sota_exp_paths) == 0:
@@ -48,6 +50,55 @@ def get_final_sota_exp(log_path: Path):
return final_sota_exp
# @st.cache_data(persist=True)
def get_sota_exp_stat(log_path: Path):
return None
trace_paths = [i for i in log_path.rglob(f"**/trace/**/*.pkl")]
if len(trace_paths) == 0:
return None
final_trace_path = max(trace_paths, key=lambda x: int(re.match(r".*Loop_(\d+).*", str(x))[1]))
with final_trace_path.open("rb") as f:
final_trace = pickle.load(f)
if hasattr(final_trace, "sota_exp_to_submit"):
st.toast("Using sota_exp_to_submit")
sota_exp = final_trace.sota_exp_to_submit
else:
sota_exp = final_trace.sota_experiment()
if sota_exp is None:
return None
sota_loop_id = None
for i, ef in enumerate(final_trace.hist):
if ef[0] == sota_exp:
sota_loop_id = i
break
sota_mle_score_paths = [i for i in log_path.rglob(f"Loop_{sota_loop_id}/running/mle_score/**/*.pkl")]
if len(sota_mle_score_paths) == 0:
# sota exp is not evaluated by mle_score
return None
with sota_mle_score_paths[0].open("rb") as f:
sota_mle_score = extract_mle_json(pickle.load(f))
sota_exp_stat = None
if sota_mle_score: # sota exp's grade output
if sota_mle_score["gold_medal"]:
sota_exp_stat = "gold"
elif sota_mle_score["silver_medal"]:
sota_exp_stat = "silver"
elif sota_mle_score["bronze_medal"]:
sota_exp_stat = "bronze"
elif sota_mle_score["above_median"]:
sota_exp_stat = "above_median"
elif sota_mle_score["valid_submission"]:
sota_exp_stat = "valid_submission"
elif sota_mle_score["submission_exists"]:
sota_exp_stat = "made_submission"
return sota_exp_stat
# @st.cache_data(persist=True)
def get_summary_df(log_folders: list[str]) -> tuple[dict, pd.DataFrame]:
summarys = {}
@@ -98,6 +149,7 @@ def get_summary_df(log_folders: list[str]) -> tuple[dict, pd.DataFrame]:
v["sota_exp_score_valid"] = final_sota_exp.result.loc["ensemble"].iloc[0]
else:
v["sota_exp_score_valid"] = None
v["sota_exp_stat_new"] = get_sota_exp_stat(Path(lf) / k)
# 调整实验名字
if "amlt" in lf:
summary[f"{lf[lf.rfind('amlt')+5:].split('/')[0]} - {k}"] = v
@@ -127,6 +179,7 @@ def get_summary_df(log_folders: list[str]) -> tuple[dict, pd.DataFrame]:
"Any Medal",
"Best Result",
"SOTA Exp",
"SOTA Exp (_to_submit)",
"SOTA Exp Score (valid)",
"SOTA Exp Score",
"Baseline Score",
@@ -195,6 +248,7 @@ def get_summary_df(log_folders: list[str]) -> tuple[dict, pd.DataFrame]:
baseline_score = baseline_df.loc[baseline_df["competition_id"] == v["competition"], "score"].item()
base_df.loc[k, "SOTA Exp"] = v.get("sota_exp_stat", None)
base_df.loc[k, "SOTA Exp (_to_submit)"] = v["sota_exp_stat_new"]
if baseline_score is not None and v.get("sota_exp_score", None) is not None:
base_df.loc[k, "Ours - Base"] = v["sota_exp_score"] - baseline_score
base_df.loc[k, "Ours vs Base"] = compare_score(v["sota_exp_score"], baseline_score)
@@ -210,6 +264,8 @@ def get_summary_df(log_folders: list[str]) -> tuple[dict, pd.DataFrame]:
base_df.loc[k, "Medium Threshold"] = v.get("median_threshold", None)
base_df["SOTA Exp"] = base_df["SOTA Exp"].replace("", pd.NA)
base_df["SOTA Exp Score (valid)"] = base_df["SOTA Exp Score (valid)"].replace("Not Calculated", 0)
base_df = base_df.astype(
{
"Total Loops": int,
@@ -350,9 +406,7 @@ def all_summarize_win():
base_df.insert(0, "Select", True)
bt1, bt2 = st.columns(2)
if bt2.toggle("Select Lite Competitions", key="select_lite"):
base_df["Select"] = base_df["Competition"].apply(lambda x: x in LITE)
else:
base_df["Select"] = True
base_df["Select"] = base_df["Competition"].isin(LITE)
if bt1.toggle("Select Best", key="select_best"):
@@ -370,8 +424,6 @@ def all_summarize_win():
best_idxs = base_df.groupby("Competition").apply(apply_func)
base_df["Select"] = base_df.index.isin(best_idxs.values)
else:
base_df["Select"] = True
base_df = st.data_editor(
base_df.style.apply(
@@ -406,13 +458,14 @@ def all_summarize_win():
subset=[
"Best Result",
"SOTA Exp",
"SOTA Exp (_to_submit)",
"SOTA Exp Score",
"SOTA Exp Score (valid)",
],
axis=0,
),
column_config={
"Select": st.column_config.CheckboxColumn("Select", default=True, help="Stat this trace.", disabled=False),
"Select": st.column_config.CheckboxColumn("Select", help="Stat this trace.", disabled=False),
},
disabled=(col for col in base_df.columns if col not in ["Select"]),
)
@@ -458,7 +511,28 @@ def all_summarize_win():
sota_exp_stat.loc["Any Medal"] = se_counts.get("Any Medal", 0)
sota_exp_stat = sota_exp_stat / base_df.shape[0] * 100
stat_df = pd.concat([total_stat, sota_exp_stat], axis=1)
# SOTA Exp (trace.sota_exp_to_submit) 统计
se_counts_new = base_df["SOTA Exp (_to_submit)"].value_counts(dropna=True)
se_counts_new.loc["made_submission"] = se_counts_new.sum()
se_counts_new.loc["Any Medal"] = (
se_counts_new.get("gold", 0) + se_counts_new.get("silver", 0) + se_counts_new.get("bronze", 0)
)
se_counts_new.loc["above_median"] = se_counts_new.get("above_median", 0) + se_counts_new.get("Any Medal", 0)
se_counts_new.loc["valid_submission"] = se_counts_new.get("valid_submission", 0) + se_counts_new.get(
"above_median", 0
)
sota_exp_stat_new = pd.Series(index=total_stat.index, dtype=int, name="SOTA Exp (_to_submit) 统计(%)")
sota_exp_stat_new.loc["Made Submission"] = se_counts_new.get("made_submission", 0)
sota_exp_stat_new.loc["Valid Submission"] = se_counts_new.get("valid_submission", 0)
sota_exp_stat_new.loc["Above Median"] = se_counts_new.get("above_median", 0)
sota_exp_stat_new.loc["Bronze"] = se_counts_new.get("bronze", 0)
sota_exp_stat_new.loc["Silver"] = se_counts_new.get("silver", 0)
sota_exp_stat_new.loc["Gold"] = se_counts_new.get("gold", 0)
sota_exp_stat_new.loc["Any Medal"] = se_counts_new.get("Any Medal", 0)
sota_exp_stat_new = sota_exp_stat_new / base_df.shape[0] * 100
stat_df = pd.concat([total_stat, sota_exp_stat, sota_exp_stat_new], axis=1)
stat_t0, stat_t1 = st.columns(2)
with stat_t0:
st.dataframe(stat_df.round(2))
+78 -47
View File
@@ -14,6 +14,7 @@ from rdagent.app.data_science.loop import DataScienceRDLoop
from rdagent.log.mle_summary import extract_mle_json, is_valid_session
from rdagent.log.storage import FileStorage
from rdagent.utils import remove_ansi_codes
from rdagent.utils.repo.diff import generate_diff_from_dict
if "show_stdout" not in state:
state.show_stdout = False
@@ -57,7 +58,8 @@ def load_times(log_path: Path):
rdloop_obj_p = next((session_path / str(max_li)).glob(f"{max_step}_*"))
rd_times = DataScienceRDLoop.load(rdloop_obj_p, do_truncate=False).loop_trace
except:
except Exception as e:
# st.toast(f"Error loading times: {e}", icon="🟡")
rd_times = {}
return rd_times
@@ -143,16 +145,19 @@ def task_win(data):
)
def workspace_win(data, instance_id=None):
show_files = {k: v for k, v in data.file_dict.items() if "test" not in k}
def workspace_win(workspace, instance_id=None, cmp_workspace=None):
show_files = {k: v for k, v in workspace.file_dict.items() if "test" not in k}
base_key = str(data.workspace_path)
base_key = str(workspace.workspace_path)
if instance_id is not None:
base_key += f"_{instance_id}"
unique_key = hashlib.md5(base_key.encode()).hexdigest()
if len(show_files) > 0:
with st.expander(f"Files in :blue[{replace_ep_path(data.workspace_path)}]"):
if cmp_workspace:
diff = generate_diff_from_dict(cmp_workspace.file_dict, show_files, "main.py")
with st.expander(":violet[**Diff with last SOTA**]"):
st.code("".join(diff), language="diff", wrap_lines=True, line_numbers=True)
with st.expander(f"Files in :blue[{replace_ep_path(workspace.workspace_path)}]"):
code_tabs = st.tabs(show_files.keys())
for ct, codename in zip(code_tabs, show_files.keys()):
with ct:
@@ -172,13 +177,13 @@ def workspace_win(data, instance_id=None):
else:
target_folder_path = Path(target_folder)
target_folder_path.mkdir(parents=True, exist_ok=True)
for filename, content in data.file_dict.items():
for filename, content in workspace.file_dict.items():
save_path = target_folder_path / filename
save_path.parent.mkdir(parents=True, exist_ok=True)
save_path.write_text(content, encoding="utf-8")
st.success(f"All files saved to: {target_folder}")
else:
st.markdown(f"No files in :blue[{replace_ep_path(data.workspace_path)}]")
st.markdown(f"No files in :blue[{replace_ep_path(workspace.workspace_path)}]")
# Helper functions
@@ -195,7 +200,9 @@ def show_text(text, lang=None):
def highlight_prompts_uri(uri):
"""高亮 URI 的格式"""
parts = uri.split(":")
return f"**{parts[0]}:**:green[**{parts[1]}**]"
if len(parts) > 1:
return f"**{parts[0]}:**:green[**{parts[1]}**]"
return f"**{uri}**"
def llm_log_win(llm_d: list):
@@ -252,22 +259,25 @@ def llm_log_win(llm_d: list):
show_text(system or "No system prompt available")
def hypothesis_win(data):
st.code(str(data).replace("\n", "\n\n"), wrap_lines=True)
def hypothesis_win(hypo):
try:
st.code(str(hypo).replace("\n", "\n\n"), wrap_lines=True)
except Exception as e:
st.write(hypo.__dict__)
def exp_gen_win(data, llm_data=None):
def exp_gen_win(exp_gen_data, llm_data=None):
st.header("Exp Gen", divider="blue", anchor="exp-gen")
if state.show_llm_log and llm_data is not None:
llm_log_win(llm_data["no_tag"])
st.subheader("Hypothesis")
hypothesis_win(data["no_tag"].hypothesis)
hypothesis_win(exp_gen_data["no_tag"].hypothesis)
st.subheader("pending_tasks")
for tasks in data["no_tag"].pending_tasks_list:
for tasks in exp_gen_data["no_tag"].pending_tasks_list:
task_win(tasks[0])
st.subheader("Exp Workspace")
workspace_win(data["no_tag"].experiment_workspace)
workspace_win(exp_gen_data["no_tag"].experiment_workspace)
def evolving_win(data, key, llm_data=None):
@@ -325,7 +335,7 @@ def coding_win(data, llm_data: dict | None = None):
workspace_win(data["no_tag"].experiment_workspace, instance_id="coding_dump")
def running_win(data, mle_score, llm_data=None):
def running_win(data, mle_score, llm_data=None, sota_exp=None):
st.header("Running", divider="blue", anchor="running")
if llm_data is not None:
common_llm_data = llm_data.pop("no_tag", [])
@@ -336,7 +346,11 @@ def running_win(data, mle_score, llm_data=None):
llm_log_win(common_llm_data)
if "no_tag" in data:
st.subheader("Exp Workspace (running final)")
workspace_win(data["no_tag"].experiment_workspace, instance_id="running_dump")
workspace_win(
data["no_tag"].experiment_workspace,
instance_id="running_dump",
cmp_workspace=sota_exp.experiment_workspace if sota_exp else None,
)
st.subheader("Result")
st.write(data["no_tag"].result)
st.subheader("MLE Submission Score" + ("" if (isinstance(mle_score, dict) and mle_score["score"]) else ""))
@@ -346,41 +360,49 @@ def running_win(data, mle_score, llm_data=None):
st.code(mle_score, wrap_lines=True)
def feedback_win(data, llm_data=None):
data = data["no_tag"]
st.header("Feedback" + ("" if bool(data) else ""), divider="orange", anchor="feedback")
def feedback_win(fb_data, llm_data=None):
fb_data = fb_data["no_tag"]
st.header("Feedback" + ("" if bool(fb_data) else ""), divider="orange", anchor="feedback")
if state.show_llm_log and llm_data is not None:
llm_log_win(llm_data["no_tag"])
st.code(str(data).replace("\n", "\n\n"), wrap_lines=True)
if data.exception is not None:
st.markdown(f"**:red[Exception]**: {data.exception}")
st.code(str(fb_data).replace("\n", "\n\n"), wrap_lines=True)
if fb_data.exception is not None:
st.markdown(f"**:red[Exception]**: {fb_data.exception}")
def sota_win(data):
def sota_win(sota_exp, trace):
st.header("SOTA Experiment", divider="rainbow", anchor="sota-exp")
if data:
if hasattr(trace, "sota_exp_to_submit") and trace.sota_exp_to_submit is not None:
st.markdown(":orange[trace.**sota_exp_to_submit**]")
sota_exp = trace.sota_exp_to_submit
else:
st.markdown(":orange[trace.**sota_experiment()**]")
if sota_exp:
st.markdown(f"**SOTA Exp Hypothesis**")
hypothesis_win(data.hypothesis)
hypothesis_win(sota_exp.hypothesis)
st.markdown("**Exp Workspace**")
workspace_win(data.experiment_workspace, instance_id="sota")
workspace_win(sota_exp.experiment_workspace, instance_id="sota")
else:
st.markdown("No SOTA experiment.")
def main_win(data, llm_data=None):
exp_gen_win(data["direct_exp_gen"], llm_data["direct_exp_gen"] if llm_data else None)
if "coding" in data:
coding_win(data["coding"], llm_data["coding"] if llm_data else None)
if "running" in data:
def main_win(loop_id, llm_data=None):
loop_data = state.data[loop_id]
exp_gen_win(loop_data["direct_exp_gen"], llm_data["direct_exp_gen"] if llm_data else None)
if "coding" in loop_data:
coding_win(loop_data["coding"], llm_data["coding"] if llm_data else None)
if "running" in loop_data:
running_win(
data["running"],
data.get("mle_score", "no submission to score"),
loop_data["running"],
loop_data.get("mle_score", "no submission to score"),
llm_data=llm_data["running"] if llm_data else None,
sota_exp=state.data[loop_id - 1].get("record", {}).get("SOTA experiment", None) if loop_id >= 1 else None,
)
if "feedback" in data:
feedback_win(data["feedback"], llm_data.get("feedback", None) if llm_data else None)
if "record" in data and "SOTA experiment" in data["record"]:
sota_win(data["record"]["SOTA experiment"])
if "feedback" in loop_data:
feedback_win(loop_data["feedback"], llm_data.get("feedback", None) if llm_data else None)
if "record" in loop_data and "SOTA experiment" in loop_data["record"]:
sota_win(loop_data["record"]["SOTA experiment"], loop_data["record"]["trace"])
def replace_ep_path(p: Path):
@@ -407,7 +429,7 @@ def summarize_data():
"Running Score (valid)",
"Running Score (test)",
"Feedback",
"e-loops",
"e-loops(coding)",
"Time",
"Exp Gen",
"Coding",
@@ -485,9 +507,11 @@ def summarize_data():
if "coding" in loop_data:
if len([i for i in loop_data["coding"].keys() if isinstance(i, int)]) == 0:
df.loc[loop, "e-loops"] = 0
df.loc[loop, "e-loops(coding)"] = 0
else:
df.loc[loop, "e-loops"] = max(i for i in loop_data["coding"].keys() if isinstance(i, int)) + 1
df.loc[loop, "e-loops(coding)"] = (
max(i for i in loop_data["coding"].keys() if isinstance(i, int)) + 1
)
if "feedback" in loop_data:
df.loc[loop, "Feedback"] = "" if bool(loop_data["feedback"]["no_tag"]) else ""
else:
@@ -508,7 +532,7 @@ def summarize_data():
total_num = x.shape[0]
valid_num = x[x["Running Score (test)"] != "N/A"].shape[0]
success_num = x[x["Feedback"] == ""].shape[0]
avg_e_loops = x["e-loops"].mean()
avg_e_loops = x["e-loops(coding)"].mean()
return pd.Series(
{
"Loop Num": total_num,
@@ -516,7 +540,7 @@ def summarize_data():
"Success Loop": success_num,
"Valid Rate": round(valid_num / total_num * 100, 2),
"Success Rate": round(success_num / total_num * 100, 2),
"Avg e-loops": round(avg_e_loops, 2),
"Avg e-loops(coding)": round(avg_e_loops, 2),
}
)
@@ -524,7 +548,7 @@ def summarize_data():
# component statistics
comp_df = (
df.loc[:, ["Component", "Running Score (test)", "Feedback", "e-loops"]]
df.loc[:, ["Component", "Running Score (test)", "Feedback", "e-loops(coding)"]]
.groupby("Component")
.apply(comp_stat_func)
)
@@ -537,7 +561,7 @@ def summarize_data():
)
comp_df["Valid Rate"] = comp_df["Valid Rate"].apply(lambda x: f"{x}%")
comp_df["Success Rate"] = comp_df["Success Rate"].apply(lambda x: f"{x}%")
comp_df.loc["Total", "Avg e-loops"] = round(df["e-loops"].mean(), 2)
comp_df.loc["Total", "Avg e-loops(coding)"] = round(df["e-loops(coding)"].mean(), 2)
st2.markdown("### Component Statistics")
st2.dataframe(comp_df)
@@ -615,8 +639,15 @@ with st.sidebar:
if "log_folder" in st.query_params:
state.log_folder = Path(st.query_params["log_folder"])
state.log_folders = [str(state.log_folder)]
else:
state.log_folder = Path(st.radio(f"Select :blue[**one log folder**]", state.log_folders))
state.log_folder = Path(
st.radio(
f"Select :blue[**one log folder**]",
state.log_folders,
format_func=lambda x: x[x.rfind("amlt") + 5 :].split("/")[0],
)
)
if not state.log_folder.exists():
st.warning(f"Path {state.log_folder} does not exist!")
else:
@@ -671,4 +702,4 @@ if state.data["competition"]:
loop_id = 0
if state.show_stdout:
stdout_win(loop_id)
main_win(state.data[loop_id], state.llm_data[loop_id] if loop_id in state.llm_data else None)
main_win(loop_id, state.llm_data[loop_id] if loop_id in state.llm_data else None)
+14 -5
View File
@@ -14,20 +14,29 @@ if "log_folders" not in state:
summary_page = st.Page("ds_summary.py", title="Summary", icon="📊")
trace_page = st.Page("ds_trace.py", title="Trace", icon="📈")
aide_page = st.Page("aide.py", title="Aide", icon="🧑‍🏫")
st.set_page_config(layout="wide", page_title="RD-Agent", page_icon="🎓", initial_sidebar_state="expanded")
st.navigation([summary_page, trace_page]).run()
st.navigation([summary_page, trace_page, aide_page]).run()
def convert_log_folder_str(lf: str) -> str:
if "/" not in lf:
return f"/data/share_folder_local/amlt/{lf.strip()}/combined_logs"
return lf.strip()
# UI - Sidebar
with st.sidebar:
st.subheader("Pages", divider="rainbow")
st.page_link(summary_page, icon="📊")
st.page_link(trace_page, icon="📈")
st.page_link(aide_page, icon="🧑‍🏫")
st.subheader("Settings", divider="rainbow")
with st.form("log_folder_form", border=False):
log_folder_str = st.text_area(
"**Log Folders**(split by ';')", placeholder=state.log_folder, value=";".join(state.log_folders)
)
log_folder_str = st.text_area("**Log Folders**(split by ';')", placeholder=state.log_folder)
if st.form_submit_button("Confirm"):
state.log_folders = [folder.strip() for folder in log_folder_str.split(";") if folder.strip()]
state.log_folders = [
convert_log_folder_str(folder) for folder in log_folder_str.split(";") if folder.strip()
]
st.rerun()