From 3df94e920c95bd458cd095f848fbfe1900d7983c Mon Sep 17 00:00:00 2001 From: XianBW <36835909+XianBW@users.noreply.github.com> Date: Tue, 26 Aug 2025 18:31:21 +0800 Subject: [PATCH] chore: refine get_summary_df and sota exp stat in log utils (#1201) * add best valid selector to get sota exp stat * remove unused in summary & refine get_summary_df * fix hypothesis table bug in trace --- rdagent/log/mle_summary.py | 23 ++-- rdagent/log/ui/ds_summary.py | 38 ++----- rdagent/log/ui/ds_trace.py | 1 + rdagent/log/ui/utils.py | 204 +++++++++++++++++------------------ 4 files changed, 122 insertions(+), 144 deletions(-) diff --git a/rdagent/log/mle_summary.py b/rdagent/log/mle_summary.py index 0ad28b6c..e6a62e24 100644 --- a/rdagent/log/mle_summary.py +++ b/rdagent/log/mle_summary.py @@ -17,7 +17,8 @@ from rdagent.scenarios.data_science.test_eval import ( NoTestEvalError, get_test_eval, ) -from rdagent.scenarios.kaggle.kaggle_crawler import score_rank + +# from rdagent.scenarios.kaggle.kaggle_crawler import score_rank from rdagent.utils.workflow import LoopBase @@ -146,10 +147,10 @@ def summarize_folder(log_folder: Path, hours: int | None = None) -> None: made_submission_num += 1 if grade_output["score"] is not None: test_scores[loop_id] = grade_output["score"] - if is_mle: - _, test_ranks[loop_id] = score_rank( - stat[log_trace_path.name]["competition"], grade_output["score"] - ) + # if is_mle: + # _, test_ranks[loop_id] = score_rank( + # stat[log_trace_path.name]["competition"], grade_output["score"] + # ) if grade_output["valid_submission"]: valid_submission_num += 1 if grade_output["above_median"]: @@ -182,10 +183,10 @@ def summarize_folder(log_folder: Path, hours: int | None = None) -> None: sota_exp_stat = "made_submission" if grade_output["score"] is not None: sota_exp_score = grade_output["score"] - if is_mle: - _, sota_exp_rank = score_rank( - stat[log_trace_path.name]["competition"], grade_output["score"] - ) + # if is_mle: + # _, sota_exp_rank = score_rank( + # stat[log_trace_path.name]["competition"], grade_output["score"] + # ) stat[log_trace_path.name].update( { @@ -198,12 +199,12 @@ def summarize_folder(log_folder: Path, hours: int | None = None) -> None: "silver_num": silver_num, "gold_num": gold_num, "test_scores": test_scores, - "test_ranks": test_ranks, + # "test_ranks": test_ranks, "valid_scores": valid_scores, "success_loop_num": success_loop_num, "sota_exp_stat": sota_exp_stat, "sota_exp_score": sota_exp_score, - "sota_exp_rank": sota_exp_rank, + # "sota_exp_rank": sota_exp_rank, "bronze_threshold": bronze_threshold, "silver_threshold": silver_threshold, "gold_threshold": gold_threshold, diff --git a/rdagent/log/ui/ds_summary.py b/rdagent/log/ui/ds_summary.py index 1b955423..1c33fcc8 100755 --- a/rdagent/log/ui/ds_summary.py +++ b/rdagent/log/ui/ds_summary.py @@ -24,29 +24,6 @@ from rdagent.log.ui.utils import ( from rdagent.scenarios.kaggle.kaggle_crawler import get_metric_direction -def days_summarize_win(): - lfs1 = [re.sub(r"log\.srv\d*", "log.srv", folder) for folder in state.log_folders] - lfs2 = [re.sub(r"log\.srv\d*", "log.srv2", folder) for folder in state.log_folders] - lfs3 = [re.sub(r"log\.srv\d*", "log.srv3", folder) for folder in state.log_folders] - - _, df1 = get_summary_df(lfs1) - _, df2 = get_summary_df(lfs2) - _, df3 = get_summary_df(lfs3) - - df = pd.concat([df1, df2, df3], axis=0) - - def mean_func(x: pd.DataFrame): - numeric_cols = x.select_dtypes(include=["int", "float"]).mean() - string_cols = x.select_dtypes(include=["object"]).agg(lambda col: ", ".join(col.fillna("none").astype(str))) - return pd.concat([numeric_cols, string_cols], axis=0).reindex(x.columns).drop("Competition") - - df = df.groupby("Competition").apply(mean_func) - if st.toggle("Show Percent", key="show_percent"): - st.dataframe(percent_df(df, show_origin=False)) - else: - st.dataframe(df) - - def curves_win(summary: dict): # draw curves cbwin1, cbwin2 = st.columns(2) @@ -110,9 +87,15 @@ def all_summarize_win(): st.warning( f"summary.pkl not found in **{lf}**\n\nRun:`dotenv run -- python rdagent/log/mle_summary.py grade_summary --log_folder={lf} --hours=<>`" ) - summary, base_df = get_summary_df(selected_folders) - if not summary: - return + summary = {} + dfs = [] + for lf in selected_folders: + s, df = get_summary_df(lf) + df.index = [f"{shorten_folder_name(lf)} - {idx}" for idx in df.index] + + dfs.append(df) + summary.update({f"{shorten_folder_name(lf)} - {k}": v for k, v in s.items()}) + base_df = pd.concat(dfs) valid_rate = float(base_df.get("Valid Improve", pd.Series()).mean()) test_rate = float(base_df.get("Test Improve", pd.Series()).mean()) @@ -215,8 +198,5 @@ def all_summarize_win(): curves_win(summary) -# with st.container(border=True): -# if st.toggle("近3天平均", key="show_3days"): -# days_summarize_win() with st.container(border=True): all_summarize_win() diff --git a/rdagent/log/ui/ds_trace.py b/rdagent/log/ui/ds_trace.py index fc0eafc8..9c62fcaa 100644 --- a/rdagent/log/ui/ds_trace.py +++ b/rdagent/log/ui/ds_trace.py @@ -946,6 +946,7 @@ def summarize_win(): st.markdown("### Hypotheses Table") hypotheses_df = df.iloc[:, :8].copy() others_expanded = pd.json_normalize(hypotheses_df["Others"].fillna({})) + others_expanded.index = hypotheses_df.index hypotheses_df = hypotheses_df.drop("Others", axis=1) hypotheses_df = hypotheses_df.drop("Parent N", axis=1) diff --git a/rdagent/log/ui/utils.py b/rdagent/log/ui/utils.py index e28cbaf1..a78426fa 100644 --- a/rdagent/log/ui/utils.py +++ b/rdagent/log/ui/utils.py @@ -22,6 +22,9 @@ from rdagent.log.ui.conf import UI_SETTING from rdagent.log.utils import extract_json, extract_loopid_func_name from rdagent.oai.llm_utils import md5_hash from rdagent.scenarios.data_science.experiment.experiment import DSExperiment +from rdagent.scenarios.data_science.proposal.exp_gen.select.submit import ( + BestValidSelector, +) from rdagent.scenarios.kaggle.kaggle_crawler import get_metric_direction LITE = [ @@ -158,13 +161,28 @@ def map_stat(sota_mle_score: dict | None) -> str: return sota_exp_stat -def _get_sota_exp_stat_hash_func(log_path: Path, to_submit: bool = True) -> str: - return _log_path_hash_func(log_path) + str(to_submit) +@cache_with_pickle(_log_path_hash_func, force=True) +def get_best_report(log_path: Path) -> dict | None: + log_storage = FileStorage(log_path) + mle_reports = [extract_json(i.content) for i in log_storage.iter_msg(pattern="**/running/mle_score/*/*.pkl")] + mle_reports = [report for report in mle_reports if report is not None and not pd.isna(report["score"])] + if mle_reports: + lower_better = mle_reports[0]["is_lower_better"] + if lower_better: + mle_reports.sort(key=lambda report: report["score"]) + else: + mle_reports.sort(key=lambda report: report["score"], reverse=True) + return mle_reports[0] + return None + + +def _get_sota_exp_stat_hash_func(log_path: Path, selector: Literal["auto", "best_valid"] = "auto") -> str: + return _log_path_hash_func(log_path) + selector @cache_with_pickle(_get_sota_exp_stat_hash_func, force=True) def get_sota_exp_stat( - log_path: Path, to_submit: bool = True + log_path: Path, selector: Literal["auto", "best_valid"] = "auto" ) -> tuple[DSExperiment | None, int | None, dict | None, str | None]: """ Get the SOTA experiment and its statistics from the log path. @@ -173,8 +191,8 @@ def get_sota_exp_stat( ---------- log_path : Path Path to the experiment log directory. - to_submit : bool, default True - If True, returns sota_exp_to_submit; if False, returns common SOTA experiment. + selector : Literal["auto", "best_valid"], default "auto" + If "auto", returns sota_exp_to_submit; if "best_valid", returns sota selected by best valid score. Returns ------- @@ -192,19 +210,19 @@ def get_sota_exp_stat( log_storage = FileStorage(log_path) # get sota exp - sota_exp_list = [ - i.content for i in log_storage.iter_msg(tag=("sota_exp_to_submit" if to_submit else "SOTA experiment")) - ] - if len(sota_exp_list) == 0: - # if no sota exp found, try to find the last trace + sota_exp = None + if selector == "auto": + sota_exp_list = [i.content for i in log_storage.iter_msg(tag="sota_exp_to_submit")] + sota_exp = sota_exp_list[-1] if sota_exp_list else None + elif selector == "best_valid": trace_list = [i.content for i in log_storage.iter_msg(tag="trace")] - final_trace = trace_list[-1] if trace_list else None - if final_trace is not None: - sota_exp = final_trace.sota_exp_to_submit if to_submit else final_trace.sota_experiment(search_type="all") - else: - sota_exp = None - else: - sota_exp = sota_exp_list[-1] + if trace_list: + final_trace = trace_list[-1] + final_trace.scen.metric_direction = get_metric_direction( + final_trace.scen.competition + ) # FIXME: remove this later. + bvs = BestValidSelector() + sota_exp = bvs.get_sota_exp_to_submit(final_trace) if sota_exp is None: return None, None, None, None @@ -240,7 +258,7 @@ def _get_score_stat_hash_func(log_path: Path, sota_loop_id: int) -> str: @cache_with_pickle(_get_score_stat_hash_func, force=True) -def get_score_stat(log_path: Path, sota_loop_id: int) -> tuple[float | None, bool]: +def get_score_stat(log_path: Path, sota_loop_id: int) -> tuple[float | None, float | None, bool, float | None]: """ Get the scores before and after merge period. @@ -268,14 +286,10 @@ def get_score_stat(log_path: Path, sota_loop_id: int) -> tuple[float | None, boo test_before_merge = [] valid_after_merge = [] test_after_merge = [] - all_report = [] submit_is_merge = False - best_is_merge = False is_lower_better = False valid_improve = False test_improve = False - best_score = None - best_stat = None total_merge_loops = 0 log_storage = FileStorage(log_path) all_trace = list(log_storage.iter_msg(tag="trace")) @@ -312,20 +326,15 @@ def get_score_stat(log_path: Path, sota_loop_id: int) -> tuple[float | None, boo is_lower_better = mle_score.get("is_lower_better", False) valid_score = pd.DataFrame(exp.result).loc["ensemble"].iloc[0] - all_report.append((valid_score, mle_score)) if is_merge: valid_after_merge.append(valid_score) - test_after_merge.append(mle_score["score"]) + if mle_score["score"] is not None: + test_after_merge.append(mle_score["score"]) else: valid_before_merge.append(valid_score) - test_before_merge.append(mle_score["score"]) - - if all_report: - all_report.sort(key=lambda x: x[0]) - best_score_dict = all_report[0][1] if is_lower_better else all_report[-1][1] - best_stat = map_stat(best_score_dict) - best_score = best_score_dict["score"] + if mle_score["score"] is not None: + test_before_merge.append(mle_score["score"]) if is_lower_better: if valid_after_merge: @@ -339,7 +348,7 @@ def get_score_stat(log_path: Path, sota_loop_id: int) -> tuple[float | None, boo test_improve = not test_before_merge or max(test_after_merge) > max(test_before_merge) merge_sota_rate = 0 if not total_merge_loops else len(test_after_merge) / total_merge_loops - return best_score, best_stat, valid_improve, test_improve, submit_is_merge, merge_sota_rate + return valid_improve, test_improve, submit_is_merge, merge_sota_rate @cache_with_pickle(_log_path_hash_func, force=True) @@ -394,19 +403,17 @@ def load_times_info(log_path: Path) -> dict[int, dict[str, dict[Literal["start_t return times_info -def _log_folders_summary_hash_func(log_folders: list[str], hours: int | None = None): - hash_str = "" - for lf in log_folders: - summary_p = Path(lf) / (f"summary.pkl" if hours is None else f"summary_{hours}h.pkl") - if summary_p.exists(): - hash_str += str(summary_p) + str(summary_p.stat().st_mtime) - else: - hash_str += f"{summary_p} not exists" +def _log_folders_summary_hash_func(log_folder: str | Path, hours: int | None = None): + summary_p = Path(log_folder) / (f"summary.pkl" if hours is None else f"summary_{hours}h.pkl") + if summary_p.exists(): + hash_str = str(summary_p) + str(summary_p.stat().st_mtime) + else: + hash_str = f"{summary_p} not exists" return md5_hash(hash_str) @cache_with_pickle(_log_folders_summary_hash_func, force=True) -def get_summary_df(log_folders: list[str], hours: int | None = None) -> tuple[dict, pd.DataFrame]: +def get_summary_df(log_folder: str | Path, hours: int | None = None) -> tuple[dict, pd.DataFrame]: """Process experiment logs and generate summary DataFrame. Several key metrics that need explanation: @@ -426,79 +433,68 @@ def get_summary_df(log_folders: list[str], hours: int | None = None) -> tuple[di and overfitting risk, totally decided by LLM """ - summarys = {} - if hours is None: - sn = "summary.pkl" + log_folder = Path(log_folder) + sn = "summary.pkl" if hours is None else f"summary_{hours}h.pkl" + if (log_folder / sn).exists(): + summary: dict = pd.read_pickle(log_folder / sn) else: - sn = f"summary_{hours}h.pkl" - for lf in log_folders: - if (Path(lf) / sn).exists(): - summarys[lf] = pd.read_pickle(Path(lf) / sn) - - if len(summarys) == 0: return {}, pd.DataFrame() - summary = {} - for lf, s in summarys.items(): - for k, v in s.items(): - stdout_p = Path(lf) / f"{k}.stdout" - if stdout_p.exists(): - v["script_time"] = get_script_time(stdout_p) - else: - v["script_time"] = None + for k, v in summary.items(): + stdout_p = log_folder / f"{k}.stdout" + if stdout_p.exists(): + v["script_time"] = get_script_time(stdout_p) + else: + v["script_time"] = None - times_info = load_times_info(Path(lf) / k) + times_info = load_times_info(log_folder / k) - exp_gen_time = coding_time = running_time = timedelta() - start_times, end_times = [], [] + exp_gen_time = coding_time = running_time = timedelta() + start_times, end_times = [], [] - for loop_times in times_info.values(): - for step_name, step_time in loop_times.items(): - duration = step_time["end_time"] - step_time["start_time"] - start_times.append(step_time["start_time"]) - end_times.append(step_time["end_time"]) + for loop_times in times_info.values(): + for step_name, step_time in loop_times.items(): + duration = step_time["end_time"] - step_time["start_time"] + start_times.append(step_time["start_time"]) + end_times.append(step_time["end_time"]) - if step_name == "exp_gen": - exp_gen_time += duration - elif step_name == "coding": - coding_time += duration - elif step_name == "running": - running_time += duration + if step_name == "exp_gen": + exp_gen_time += duration + elif step_name == "coding": + coding_time += duration + elif step_name == "running": + running_time += duration - all_time = (max(end_times) - min(start_times)) if start_times else timedelta() - v["exec_time"] = str(all_time).split(".")[0] - v["exp_gen_time"] = str(exp_gen_time).split(".")[0] - v["coding_time"] = str(coding_time).split(".")[0] - v["running_time"] = str(running_time).split(".")[0] + all_time = (max(end_times) - min(start_times)) if start_times else timedelta() + v["exec_time"] = str(all_time).split(".")[0] + v["exp_gen_time"] = str(exp_gen_time).split(".")[0] + v["coding_time"] = str(coding_time).split(".")[0] + v["running_time"] = str(running_time).split(".")[0] - # overwrite sota_exp_stat in summary.pkl because it may not be correct in multi-trace - sota_exp_submit, v["sota_loop_id_new"], sota_submit_report, v["sota_exp_stat_new"] = get_sota_exp_stat( - Path(lf) / k, to_submit=True + # overwrite sota_exp_stat in summary.pkl because it may not be correct in multi-trace + sota_exp_submit, v["sota_loop_id_new"], sota_submit_report, v["sota_exp_stat_new"] = get_sota_exp_stat( + log_folder / k, selector="auto" + ) + sota_exp_bv, v["sota_loop_id"], sota_bv_report, v["sota_exp_stat"] = get_sota_exp_stat( + log_folder / k, selector="best_valid" + ) + ( + v["valid_improve"], + v["test_improve"], + v["submit_is_merge"], + v["merge_sota_rate"], + ) = get_score_stat(log_folder / k, v["sota_loop_id_new"]) + + if sota_exp_submit is not None: + try: + sota_submit_result = sota_exp_submit.result + except AttributeError: # Compatible with old versions + sota_submit_result = sota_exp_submit.__dict__["result"] + v["sota_exp_score_valid_new"] = ( + sota_submit_result.loc["ensemble"].iloc[0] if sota_submit_result is not None else None ) - ( - v["sota_exp_score"], - v["sota_exp_stat"], - v["valid_improve"], - v["test_improve"], - v["submit_is_merge"], - v["merge_sota_rate"], - ) = get_score_stat(Path(lf) / k, v["sota_loop_id_new"]) - if sota_exp_submit is not None: - try: - sota_submit_result = sota_exp_submit.result - except AttributeError: # Compatible with old versions - sota_submit_result = sota_exp_submit.__dict__["result"] - v["sota_exp_score_valid_new"] = ( - sota_submit_result.loc["ensemble"].iloc[0] if sota_submit_result is not None else None - ) - v["sota_exp_score_new"] = sota_submit_report["score"] if sota_submit_report else None - # change experiment name - if "amlt" in lf: - summary[f"{lf[lf.rfind('amlt')+5:].split('/')[0]} - {k}"] = v - elif "ep" in lf: - summary[f"{lf[lf.rfind('ep'):]} - {k}"] = v - else: - summary[f"{lf} - {k}"] = v + v["sota_exp_score"] = sota_bv_report["score"] if sota_bv_report else None + v["sota_exp_score_new"] = sota_submit_report["score"] if sota_submit_report else None summary = {k: v for k, v in summary.items() if "competition" in v} base_df = pd.DataFrame(