INNER CODE UNIT · Python
evaluate_best_run
OSU-NLP-Group/ScienceAgentBench · calculate_metrics.py:6
def evaluate_best_run(run_logs: List[List[Dict]], eval_logs: List[List[Dict]]):
selected_run = []
for i in range(len(run_logs[0])):
task_traj_all = [r[i] for r in eval_logs]
task_cost = [r[i]["cost"] for r in run_logs]
for r, c in zip(task_traj_all, task_cost):
r["cost"] = c
task_traj_all = [r[i] for r in eval_logs]
task_sr = [t["success_rate"] for t in task_traj_all]
best_sr = max(task_sr)
task_traj = [t for t in task_traj_all if t["success_rate"] == best_sr]
if len(task_traj) > 1:
task_ver = [t["valid_program"] for t in task_traj]
best_ver = max(task_ver)