Skip to content

Commit f6de20d

Browse files
committed
feat : better run_notes
1 parent ff559de commit f6de20d

6 files changed

Lines changed: 180 additions & 121 deletions

File tree

sources/cache/openrouter_pricing.json

Lines changed: 35 additions & 51 deletions
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,26 @@
11
{
2-
"timestamp": "2026-03-09T15:48:31.473480",
2+
"timestamp": "2026-03-12T10:02:59.367821",
33
"pricing": {
4+
"openrouter/hunter-alpha": {
5+
"input": 0.0,
6+
"output": 0.0
7+
},
8+
"openrouter/healer-alpha": {
9+
"input": 0.0,
10+
"output": 0.0
11+
},
12+
"nvidia/nemotron-3-super-120b-a12b:free": {
13+
"input": 0.0,
14+
"output": 0.0
15+
},
16+
"bytedance-seed/seed-2.0-lite": {
17+
"input": 0.25,
18+
"output": 2.0
19+
},
20+
"qwen/qwen3.5-9b": {
21+
"input": 0.09999999999999999,
22+
"output": 0.15
23+
},
424
"openai/gpt-5.4-pro": {
525
"input": 30.0,
626
"output": 180.0
@@ -78,12 +98,12 @@
7898
"output": 2.34
7999
},
80100
"minimax/minimax-m2.5": {
81-
"input": 0.295,
82-
"output": 1.2
101+
"input": 0.27,
102+
"output": 0.95
83103
},
84104
"z-ai/glm-5": {
85-
"input": 0.7999999999999999,
86-
"output": 2.56
105+
"input": 0.72,
106+
"output": 2.3
87107
},
88108
"qwen/qwen3-max-thinking": {
89109
"input": 0.78,
@@ -274,8 +294,8 @@
274294
"output": 1.2
275295
},
276296
"deepseek/deepseek-v3.2": {
277-
"input": 0.25,
278-
"output": 0.39999999999999997
297+
"input": 0.26,
298+
"output": 0.38
279299
},
280300
"prime-intellect/intellect-3": {
281301
"input": 0.19999999999999998,
@@ -422,8 +442,8 @@
422442
"output": 2.5
423443
},
424444
"qwen/qwen3-vl-30b-a3b-thinking": {
425-
"input": 0.0,
426-
"output": 0.0
445+
"input": 0.13,
446+
"output": 1.56
427447
},
428448
"qwen/qwen3-vl-30b-a3b-instruct": {
429449
"input": 0.13,
@@ -437,10 +457,6 @@
437457
"input": 0.39,
438458
"output": 1.9
439459
},
440-
"z-ai/glm-4.6:exacto": {
441-
"input": 0.44,
442-
"output": 1.76
443-
},
444460
"anthropic/claude-sonnet-4.5": {
445461
"input": 3.0,
446462
"output": 15.0
@@ -462,8 +478,8 @@
462478
"output": 0.39999999999999997
463479
},
464480
"qwen/qwen3-vl-235b-a22b-thinking": {
465-
"input": 0.0,
466-
"output": 0.0
481+
"input": 0.26,
482+
"output": 2.6
467483
},
468484
"qwen/qwen3-vl-235b-a22b-instruct": {
469485
"input": 0.19999999999999998,
@@ -481,10 +497,6 @@
481497
"input": 1.25,
482498
"output": 10.0
483499
},
484-
"deepseek/deepseek-v3.1-terminus:exacto": {
485-
"input": 0.21,
486-
"output": 0.7899999999999999
487-
},
488500
"deepseek/deepseek-v3.1-terminus": {
489501
"input": 0.21,
490502
"output": 0.7899999999999999
@@ -502,8 +514,8 @@
502514
"output": 0.975
503515
},
504516
"qwen/qwen3-next-80b-a3b-thinking": {
505-
"input": 0.15,
506-
"output": 1.2
517+
"input": 0.0975,
518+
"output": 0.78
507519
},
508520
"qwen/qwen3-next-80b-a3b-instruct:free": {
509521
"input": 0.0,
@@ -537,10 +549,6 @@
537549
"input": 0.39999999999999997,
538550
"output": 2.0
539551
},
540-
"moonshotai/kimi-k2-0905:exacto": {
541-
"input": 0.6,
542-
"output": 2.5
543-
},
544552
"qwen/qwen3-30b-a3b-thinking-2507": {
545553
"input": 0.051,
546554
"output": 0.33999999999999997
@@ -609,10 +617,6 @@
609617
"input": 0.039,
610618
"output": 0.19
611619
},
612-
"openai/gpt-oss-120b:exacto": {
613-
"input": 0.039,
614-
"output": 0.19
615-
},
616620
"openai/gpt-oss-20b:free": {
617621
"input": 0.0,
618622
"output": 0.0
@@ -665,10 +669,6 @@
665669
"input": 0.22,
666670
"output": 1.0
667671
},
668-
"qwen/qwen3-coder:exacto": {
669-
"input": 0.22,
670-
"output": 1.7999999999999998
671-
},
672672
"bytedance/ui-tars-1.5-7b": {
673673
"input": 0.09999999999999999,
674674
"output": 0.19999999999999998
@@ -1133,10 +1133,6 @@
11331133
"input": 0.2,
11341134
"output": 0.2
11351135
},
1136-
"raifle/sorcererlm-8x22b": {
1137-
"input": 4.5,
1138-
"output": 4.5
1139-
},
11401136
"thedrummer/unslopnemo-12b": {
11411137
"input": 0.39999999999999997,
11421138
"output": 0.39999999999999997
@@ -1193,10 +1189,6 @@
11931189
"input": 0.12,
11941190
"output": 0.39
11951191
},
1196-
"neversleep/llama-3.1-lumimaid-8b": {
1197-
"input": 0.09,
1198-
"output": 0.6
1199-
},
12001192
"cohere/command-r-08-2024": {
12011193
"input": 0.15,
12021194
"output": 0.6
@@ -1206,8 +1198,8 @@
12061198
"output": 10.0
12071199
},
12081200
"sao10k/l3.1-euryale-70b": {
1209-
"input": 0.65,
1210-
"output": 0.75
1201+
"input": 0.85,
1202+
"output": 0.85
12111203
},
12121204
"qwen/qwen-2.5-vl-7b-instruct": {
12131205
"input": 0.2,
@@ -1277,10 +1269,6 @@
12771269
"input": 0.14,
12781270
"output": 0.14
12791271
},
1280-
"meta-llama/llama-guard-2-8b": {
1281-
"input": 0.19999999999999998,
1282-
"output": 0.19999999999999998
1283-
},
12841272
"openai/gpt-4o-2024-05-13": {
12851273
"input": 5.0,
12861274
"output": 15.0
@@ -1333,10 +1321,6 @@
13331321
"input": 0.54,
13341322
"output": 0.54
13351323
},
1336-
"neversleep/noromaid-20b": {
1337-
"input": 1.0,
1338-
"output": 1.75
1339-
},
13401324
"alpindale/goliath-120b": {
13411325
"input": 3.75,
13421326
"output": 7.5

sources/core/dgm.py

Lines changed: 3 additions & 16 deletions
Original file line numberDiff line numberDiff line change
@@ -281,18 +281,7 @@ async def start_dgm(
281281

282282
rewards_history = []
283283
assertion_history = [] # Track [passed, total] per iteration
284-
285-
if self.process_id is None and max_iteration > 1:
286-
print("Setup reward visualization.")
287-
self.viz_utils.create_rewards_curve_plot(goal)
288-
elif scenario_rubric and judge:
289-
print("Setup scenario visualization.")
290-
scenario = ScenarioLoader().load_scenario(scenario_rubric)
291-
if scenario:
292-
total_assertions = len(scenario.get("assertions", []))
293-
self.viz_utils.create_assertion_progress_plot(
294-
scenario_rubric, total_assertions
295-
)
284+
self.viz_utils.create_rewards_curve_plot(goal)
296285

297286
run0 = IndividualRun(
298287
goal=goal,
@@ -499,9 +488,7 @@ async def _evaluate_and_calculate_cost(
499488
if judge and uuid:
500489
answer = answer if executed else "workflow failed to execute."
501490
eval_type = await self._evaluate_workflow(uuid, answer, scenario_rubric, assertion_history)
502-
503491
# Calculate cost regardless of execution success
504-
# This includes workflow generation LLM costs even when execution fails
505492
cost_start = time.time()
506493
exec_cost = self.pricing.calculate_cost(uuid)
507494
cost_time = time.time() - cost_start
@@ -534,8 +521,8 @@ async def _evaluate_workflow(
534521

535522
def _update_assertion_history(self, eval_result: dict, assertion_history: list):
536523
"""Update assertion history with evaluation results."""
537-
passed = eval_result.get('passed_assertions', 0)
538-
total = eval_result.get('total_assertions', 0)
524+
passed = eval_result.get('passed_assertions', eval_result.get('earned_points', 0))
525+
total = eval_result.get('total_assertions', eval_result.get('total_points', 100))
539526
assertion_history.append([passed, total])
540527
print(f"\033[94m📊 Assertions Progress: {passed}/{total} "
541528
f"({passed/total*100 if total > 0 else 0:.0f}%)\033[0m")

sources/evaluation/csv_mode.py

Lines changed: 16 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -187,13 +187,24 @@ def _save_run_notes(self, capsule_name: str, goal: str,
187187
"total_eval": len(sab_runs)
188188
}
189189
if sab_runs:
190+
runs_data = sab_runs[-1].get('runs', [])
190191
notes = {
191192
**notes,
193+
"capsule_name": capsule_name,
192194
"ver_success": sum(1 for run in sab_runs if run.get('VER', False)),
193195
"sr_success": sum(1 for run in sab_runs if run.get('SR', False)),
194196
"avg_cbs": sum(run.get('CBS', 0.0) for run in sab_runs) / len(sab_runs),
195197
"total_cost": sum(run.get('eval_cost', 0.0) for run in sab_runs),
196-
"is_success": sab_runs[-1].get('SR', False)
198+
"is_success": sab_runs[-1].get('SR', False),
199+
"task_cost": sab_runs[-1].get('eval_cost', 0.0),
200+
"max_judge_reward": max((getattr(run, 'reward', 0.0) for run in runs_data), default=0.0),
201+
"evolution_iterations": len(runs_data),
202+
"evolved_workflows_uuids": [getattr(run, 'current_uuid', '') for run in runs_data],
203+
"evolution_rewards": [getattr(run, 'reward', 0.0) for run in runs_data],
204+
"evolution_costs": [getattr(run, 'cost', 0.0) for run in runs_data],
205+
"evolution_total_cost": sum(getattr(run, 'cost', 0.0) for run in runs_data),
206+
"evolution_avg_reward": sum(getattr(run, 'reward', 0.0) for run in runs_data) / len(runs_data) if runs_data else 0,
207+
"evolution_avg_cost": sum(getattr(run, 'cost', 0.0) for run in runs_data) / len(runs_data) if runs_data else 0
197208
}
198209

199210
notes_file = self.run_notes_dir / f"{capsule_name}.json"
@@ -345,7 +356,8 @@ def _evaluate_with_science_agent_bench(
345356
'SR': eval_results['SR'][0],
346357
'SR_message': eval_results['SR'][1],
347358
'CBS': eval_results['CBS'],
348-
'eval_cost': eval_results['cost']
359+
'eval_cost': eval_results['cost'],
360+
'runs': runs
349361
})
350362
print(f"\033[95m{eval_results['summary']}\033[0m")
351363

@@ -364,7 +376,8 @@ def _evaluate_with_science_agent_bench(
364376
'VER': False,
365377
'SR': False,
366378
'CBS': 0.0,
367-
'eval_error': str(eval_error)
379+
'eval_error': str(eval_error),
380+
'runs': runs
368381
})
369382

370383
return execution_data

0 commit comments

Comments
 (0)