Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
The table of contents is too big for display.
Diff view
Diff view
  •  
  •  
  •  
Original file line number Diff line number Diff line change
@@ -0,0 +1,13 @@
# Actava · openai-agents · nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16:peft:262144

Submitted: 2026-08-12 · chi-bench chi-bench-v1.0.0 · pass@1: **0.0%**

| Domain | pass@1 | n_trials |
|---|---|---|
| pa_provider | 0.0% | 25 |
| pa_um | 0.0% | 25 |
| cm | 0.0% | 25 |

Run executed 2026-07-22 (`nemotron3_ultra_tinker_full_2026_07 experiment matrix`).

See `submission.json` for the full manifest, `provenance.json` for reproducibility info.
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
{
"chi_bench_git_sha": "2395e9e2ae7d42cdf2750dc3e67daeb0e672b910",
"image_digest": null,
"judge_model": "claude-opus-4-7",
"harness_version": "0.1.0",
"code_dirty": false,
"dataset_version": "chi-bench-v1.0.0",
"environment": "modal",
"started_at": "2026-07-22T22:31:29.302007Z",
"finished_at": "2026-07-22T23:37:41.111758Z",
"source": "nemotron3_ultra_tinker_full_2026_07 experiment matrix",
"cost_basis": "74 of 75 trials carry price-table-estimated cost; 1 trial reports no usage. Rate reflects the limited-time Tinker discount verified 2026-07-22.",
"eval_validity": "Closed per docs/superpowers/plans/2026-07-22-nemotron-tinker-eval.md: 3/3 canary, 75/75 trials and scorecards, no infrastructure exceptions.",
"trajectory_repair": {
"count": 1,
"reason": "1 pa_provider trial (MaxTurnsExceeded) completed verification but emitted no trace.jsonl.",
"method": "Generated ATIF-v1.2 error-only trajectories with the official openai_agents_harness._build_atif_trajectory(records=[], ...) builder from each preserved agent/run_result.json; original logs were not modified."
}
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
benchmark,dataset_version,submission_id,team,agent,model,domain,pass_at_1,n_trials,n_tasks,mean_cost_usd,mean_walltime_s,submitted_at
chi-bench,chi-bench-v1.0.0,nemotron-3-ultra-openai-agents,Actava,openai-agents,nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16:peft:262144,overall,0.0,75,75,0.47223568448,91.86829674666667,2026-08-12T20:20:00Z
chi-bench,chi-bench-v1.0.0,nemotron-3-ultra-openai-agents,Actava,openai-agents,nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16:peft:262144,pa_provider,0.0,25,25,0.39578937776000006,85.60383688,2026-08-12T20:20:00Z
chi-bench,chi-bench-v1.0.0,nemotron-3-ultra-openai-agents,Actava,openai-agents,nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16:peft:262144,pa_um,0.0,25,25,0.45005764624,62.04124316,2026-08-12T20:20:00Z
chi-bench,chi-bench-v1.0.0,nemotron-3-ultra-openai-agents,Actava,openai-agents,nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16:peft:262144,cm,0.0,25,25,0.57086002944,127.9598102,2026-08-12T20:20:00Z
Original file line number Diff line number Diff line change
@@ -0,0 +1,30 @@
schema: chi-bench/submission/v1
submission:
id: nemotron-3-ultra-openai-agents
team: Actava
contact: dark.savi@gmail.com
agent: openai-agents
model: nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16:peft:262144
notes: 'Nemotron 3 Ultra 256K via the OpenAI Agents SDK harness on Tinker Chat Completions.

'
agent_kwargs:
provider_route: tinker
reasoning_effort: high
max_turns: '50'
max_retries: '10'
max_tool_return_chars: '100000'
run:
environment: modal
n_attempts: 1
concurrency: 5
max_retries: 2
timeout_multiplier: 1.0
agent_timeout_multiplier: 2.0
env_file: .env
dataset:
version: chi-bench-v1.0.0
domains:
- pa_provider
- pa_um
- cm
Original file line number Diff line number Diff line change
@@ -0,0 +1,74 @@
{
"schema": "chi-bench/submission/v1",
"submission": {
"id": "nemotron-3-ultra-openai-agents",
"team": "Actava",
"contact": "dark.savi@gmail.com",
"agent": "openai-agents",
"model": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16:peft:262144",
"notes": "NVIDIA Nemotron 3 Ultra 256K via the OpenAI Agents SDK harness on Tinker Chat Completions with an explicit tinker provider route. Scored 0/75 binary pass@1 with mean fractional reward 0.217, i.e. partial task progress that never met a full rubric. Infrastructure was healthy: 75/75 verifier-backed results and scorecards, preflight and canary gates passed, and the single agent exception was a genuine MaxTurnsExceeded outcome rather than a harness fault.\n",
"submitted_at": "2026-08-12T20:20:00Z"
},
"dataset": {
"name": "chi-bench",
"version": "chi-bench-v1.0.0",
"domains": [
"pa_provider",
"pa_um",
"cm"
]
},
"results": {
"overall": {
"n_trials": 75,
"n_tasks": 75,
"pass_at_1": 0.0,
"mean_cost_usd": 0.47223568448,
"mean_walltime_s": 91.86829674666667
},
"per_domain": {
"pa_provider": {
"n_trials": 25,
"n_tasks": 25,
"pass_at_1": 0.0,
"mean_cost_usd": 0.39578937776000006,
"mean_walltime_s": 85.60383688
},
"pa_um": {
"n_trials": 25,
"n_tasks": 25,
"pass_at_1": 0.0,
"mean_cost_usd": 0.45005764624,
"mean_walltime_s": 62.04124316
},
"cm": {
"n_trials": 25,
"n_tasks": 25,
"pass_at_1": 0.0,
"mean_cost_usd": 0.57086002944,
"mean_walltime_s": 127.9598102
}
},
"mean_cost_usd": 0.47223568448,
"mean_walltime_s": 91.86829674666667
},
"provenance": {
"chi_bench_git_sha": "2395e9e2ae7d42cdf2750dc3e67daeb0e672b910",
"image_digest": null,
"judge_model": "claude-opus-4-7",
"harness_version": "0.1.0",
"code_dirty": false,
"dataset_version": "chi-bench-v1.0.0",
"environment": "modal",
"started_at": "2026-07-22T22:31:29.302007Z",
"finished_at": "2026-07-22T23:37:41.111758Z",
"source": "nemotron3_ultra_tinker_full_2026_07 experiment matrix",
"cost_basis": "74 of 75 trials carry price-table-estimated cost; 1 trial reports no usage. Rate reflects the limited-time Tinker discount verified 2026-07-22.",
"eval_validity": "Closed per docs/superpowers/plans/2026-07-22-nemotron-tinker-eval.md: 3/3 canary, 75/75 trials and scorecards, no infrastructure exceptions.",
"trajectory_repair": {
"count": 1,
"reason": "1 pa_provider trial (MaxTurnsExceeded) completed verification but emitted no trace.jsonl.",
"method": "Generated ATIF-v1.2 error-only trajectories with the official openai_agents_harness._build_atif_trajectory(records=[], ...) builder from each preserved agent/run_result.json; original logs were not modified."
}
}
}
Binary file not shown.
Original file line number Diff line number Diff line change
@@ -0,0 +1,117 @@
{
"id": "63dd5189-7af6-4840-958c-3112bb7246c4",
"task_name": "actava-ai/cm_afib_moderate_anxious_001",
"trial_name": "cm_afib_moderate_anxious_001__eYBptGR",
"trial_uri": "file:///Users/haolin.chen/.config/superpowers/worktrees/chi-bench/frontier-models-smoke/logs/experiments/nemotron3_ultra_tinker_full_2026_07/01_openai-agents_nvidia-NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16-peft-262144_cm/2026-07-22__16-14-44/cm_afib_moderate_anxious_001__eYBptGR",
"task_id": {
"path": "/var/folders/51/f74lzb_9073dvd740hb4qmd80000gn/T/chi_bench-modal-9xx7v_gy/cm_afib_moderate_anxious_001"
},
"source": "chi_bench-modal-9xx7v_gy",
"task_checksum": "ee0491f42e269fa02a2442ddbb574b37d6673ccd3a6c8fbcd403c3d2ff8eccfa",
"config": {
"task": {
"path": "/var/folders/51/f74lzb_9073dvd740hb4qmd80000gn/T/chi_bench-modal-9xx7v_gy/cm_afib_moderate_anxious_001",
"git_url": null,
"git_commit_id": null,
"name": null,
"ref": null,
"overwrite": false,
"download_dir": null,
"source": "chi_bench-modal-9xx7v_gy"
},
"trial_name": "cm_afib_moderate_anxious_001__eYBptGR",
"trials_dir": "logs/experiments/nemotron3_ultra_tinker_full_2026_07/01_openai-agents_nvidia-NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16-peft-262144_cm/2026-07-22__16-14-44",
"timeout_multiplier": 1.0,
"agent_timeout_multiplier": 2.0,
"verifier_timeout_multiplier": null,
"agent_setup_timeout_multiplier": null,
"environment_build_timeout_multiplier": null,
"agent": {
"name": null,
"import_path": "chi_bench.experiment.agents.openai_agents_harness:OpenAIAgentsHarness",
"model_name": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16:peft:262144",
"override_timeout_sec": null,
"override_setup_timeout_sec": null,
"max_timeout_sec": null,
"kwargs": {
"api_mode": "chat_completions",
"max_retries": 10,
"max_tool_return_chars": 100000,
"max_turns": 50,
"provider_route": "tinker"
},
"env": {
"OPENAI_API_KEY": "${OPENAI_API_KEY}",
"ANTHROPIC_API_KEY": "${ANTHROPIC_API_KEY}",
"GEMINI_API_KEY": "${GEMINI_API_KEY}",
"OPENROUTER_API_KEY": "${OPENROUTER_API_KEY}",
"TINKER_API_KEY": "${TINKER_API_KEY}"
}
},
"environment": {
"type": null,
"import_path": "chi_bench.experiment.modal_env:ChiBenchModalEnvironment",
"force_build": false,
"delete": true,
"override_cpus": null,
"override_memory_mb": null,
"override_storage_mb": null,
"override_gpus": null,
"suppress_override_warnings": false,
"mounts_json": null,
"env": {},
"kwargs": {
"sandbox_timeout_secs": 86400
}
},
"verifier": {
"override_timeout_sec": null,
"max_timeout_sec": null,
"env": {},
"disable": false
},
"artifacts": [],
"job_id": "0968fe18-aa0c-4920-9ca6-521a2348e762"
},
"agent_info": {
"name": "openai-agents",
"version": "0.13.6",
"model_info": {
"name": "NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16:peft:262144",
"provider": "nvidia"
}
},
"agent_result": {
"n_input_tokens": 2789690,
"n_cache_tokens": 2660352,
"n_output_tokens": 27654,
"cost_usd": null,
"rollout_details": null,
"metadata": null
},
"verifier_result": {
"rewards": {
"reward": 0.0
}
},
"exception_info": null,
"started_at": "2026-07-22T23:27:25.892612Z",
"finished_at": "2026-07-22T23:40:22.055342Z",
"environment_setup": {
"started_at": "2026-07-22T23:27:25.896922Z",
"finished_at": "2026-07-22T23:28:09.419297Z"
},
"agent_setup": {
"started_at": "2026-07-22T23:28:09.419308Z",
"finished_at": "2026-07-22T23:28:13.514756Z"
},
"agent_execution": {
"started_at": "2026-07-22T23:28:13.514794Z",
"finished_at": "2026-07-22T23:36:14.038666Z"
},
"verifier": {
"started_at": "2026-07-22T23:36:17.965201Z",
"finished_at": "2026-07-22T23:40:20.501063Z"
},
"step_results": null
}
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
{"reward": 0.0}
Original file line number Diff line number Diff line change
@@ -0,0 +1,116 @@
{
"binary_reward": 0.0,
"fractional_reward": 0.8947368421052632,
"passed_checks": 17,
"total_checks": 19,
"check_scores": {
"cm.assessment.record_exists": 1.0,
"cm.assessment.completed": 1.0,
"cm.assessment.required_sections_present": 1.0,
"cm.care_plan.record_exists": 1.0,
"cm.care_plan.finalized": 1.0,
"cm.care_plan.problem_count": 1.0,
"cm.care_plan.goal_structure": 1.0,
"cm.care_plan.intervention_structure": 1.0,
"cm.care_plan.escalation_conditions_present": 1.0,
"cm.care_plan.follow_up_cadence_present": 1.0,
"cm.chart_review.record_exists": 1.0,
"cm.cross_stage.target_status": 1.0,
"cm.cross_stage.audit_actions": 1.0,
"cm.cross_stage.no_forbidden_mutations": 1.0,
"judge.cm.chart_review.quality": 0.0,
"judge.cm.outreach.quality": 0.0,
"judge.cm.assessment.quality": 1.0,
"judge.cm.care_plan.quality": 1.0,
"judge.cm.stage_coherence": 1.0
},
"checks": {
"cm.assessment.record_exists": true,
"cm.assessment.completed": true,
"cm.assessment.required_sections_present": true,
"cm.care_plan.record_exists": true,
"cm.care_plan.finalized": true,
"cm.care_plan.problem_count": true,
"cm.care_plan.goal_structure": true,
"cm.care_plan.intervention_structure": true,
"cm.care_plan.escalation_conditions_present": true,
"cm.care_plan.follow_up_cadence_present": true,
"cm.chart_review.record_exists": true,
"cm.cross_stage.target_status": true,
"cm.cross_stage.audit_actions": true,
"cm.cross_stage.no_forbidden_mutations": true,
"judge.cm.chart_review.quality": false,
"judge.cm.outreach.quality": false,
"judge.cm.assessment.quality": true,
"judge.cm.care_plan.quality": true,
"judge.cm.stage_coherence": true
},
"failed_checks": [
"judge.cm.chart_review.quality",
"judge.cm.outreach.quality"
],
"not_applicable_checks": [],
"stages": {
"cm_assessment": {
"passed": true,
"checks": {
"cm.assessment.record_exists": true,
"cm.assessment.completed": true,
"cm.assessment.required_sections_present": true
},
"passed_count": 3,
"total_count": 3,
"not_applicable_count": 0,
"details": {
"case_id": "CM-CASE-CM_AFIB_MODERATE_ANXIOUS_001",
"record_count": 1
}
},
"cm_care_plan": {
"passed": true,
"checks": {
"cm.care_plan.record_exists": true,
"cm.care_plan.finalized": true,
"cm.care_plan.problem_count": true,
"cm.care_plan.goal_structure": true,
"cm.care_plan.intervention_structure": true,
"cm.care_plan.escalation_conditions_present": true,
"cm.care_plan.follow_up_cadence_present": true
},
"passed_count": 7,
"total_count": 7,
"not_applicable_count": 0,
"details": {
"case_id": "CM-CASE-CM_AFIB_MODERATE_ANXIOUS_001",
"record_count": 1
}
},
"cm_chart_review": {
"passed": true,
"checks": {
"cm.chart_review.record_exists": true
},
"passed_count": 1,
"total_count": 1,
"not_applicable_count": 0,
"details": {
"case_id": "CM-CASE-CM_AFIB_MODERATE_ANXIOUS_001",
"record_count": 1
}
},
"cm_cross_stage": {
"passed": true,
"checks": {
"cm.cross_stage.target_status": true,
"cm.cross_stage.audit_actions": true,
"cm.cross_stage.no_forbidden_mutations": true
},
"passed_count": 3,
"total_count": 3,
"not_applicable_count": 0,
"details": {
"case_id": "CM-CASE-CM_AFIB_MODERATE_ANXIOUS_001"
}
}
}
}
Binary file not shown.
Loading
Loading