Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
The table of contents is too big for display.
Diff view
Diff view
  •  
  •  
  •  
Original file line number Diff line number Diff line change
@@ -0,0 +1,22 @@
# chi-Bench submission

- Submission ID: `u2med-u2medfellow-zzq`
- Agent: `U2MedFellow`
- Model: `U2Med`
- Dataset: `chi-bench-v1.0.0`
- Submitted at: `2026-09-03T03:01:09Z`

## Results

| Domain | Pass@1 | Trials | Tasks |
|---|---:|---:|---:|
| Overall | 0.586667 | 75 | 75 |
| PA Provider | 0.520000 | 25 | 25 |
| PA UM | 0.520000 | 25 | 25 |
| CM | 0.720000 | 25 | 25 |

## Selection

The packet integrates the results for all three benchmark domains in the current submission directory.

The packet uses the public identity `model=U2Med`, `agent=U2MedFellow`. Trial evidence and benchmark results are included in the submission packet.
Original file line number Diff line number Diff line change
@@ -0,0 +1,9 @@
{
"chi_bench_git_sha": null,
"image_digest": null,
"judge_model": "claude-opus-4-7",
"harness_version": "0.1.0",
"selection": {
"description": "Integrated results for all three benchmark domains in the current submission directory."
}
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
benchmark,dataset_version,submission_id,team,agent,model,domain,pass_at_1,n_trials,n_tasks,mean_cost_usd,mean_walltime_s,submitted_at
chi-bench,chi-bench-v1.0.0,u2med-u2medfellow-zzq,U2Med,U2MedFellow,U2Med,overall,0.5866666666666667,75,75,0.0,0.0,2026-09-03T03:01:09Z
chi-bench,chi-bench-v1.0.0,u2med-u2medfellow-zzq,U2Med,U2MedFellow,U2Med,pa_provider,0.52,25,25,0.0,0.0,2026-09-03T03:01:09Z
chi-bench,chi-bench-v1.0.0,u2med-u2medfellow-zzq,U2Med,U2MedFellow,U2Med,pa_um,0.52,25,25,0.0,0.0,2026-09-03T03:01:09Z
chi-bench,chi-bench-v1.0.0,u2med-u2medfellow-zzq,U2Med,U2MedFellow,U2Med,cm,0.72,25,25,0.0,0.0,2026-09-03T03:01:09Z
Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
schema: chi-bench/submission/v1

submission:
id: u2med-u2medfellow-zzq
team: U2Med
contact: zhongzhiqiang-intern@example.com
agent: U2MedFellow
model: U2Med
notes: |
Leaderboard packet integrates the results for all three benchmark domains
in the current submission directory. The public leaderboard identity is
model U2Med with agent U2MedFellow. Trial evidence and benchmark results
are included in the submission packet.

run:
environment: docker
n_attempts: 1
concurrency: 4
max_retries: 0
timeout_multiplier: 2.0
env_file: .env

dataset:
version: chi-bench-v1.0.0
domains: [pa_provider, pa_um, cm]

paths:
data_root: data
output_root: logs/submissions/2026-09-03-u2med-u2medfellow-zzq
Original file line number Diff line number Diff line change
@@ -0,0 +1,64 @@
{
"schema": "chi-bench/submission/v1",
"submission": {
"id": "u2med-u2medfellow-zzq",
"team": "U2Med",
"contact": "zhongzhiqiang-intern@example.com",
"agent": "U2MedFellow",
"model": "U2Med",
"notes": "Leaderboard packet integrates the results for all three benchmark domains in the current submission directory. The leaderboard identity is model U2Med with agent U2MedFellow. Trial evidence and benchmark results are included in the submission packet.",
"submitted_at": "2026-09-03T03:01:09Z"
},
"dataset": {
"name": "chi-bench",
"version": "chi-bench-v1.0.0",
"domains": [
"pa_provider",
"pa_um",
"cm"
]
},
"results": {
"overall": {
"pass_at_1": 0.5866666666666667,
"n_trials": 75,
"n_tasks": 75,
"mean_cost_usd": 0.0,
"mean_walltime_s": 0.0
},
"per_domain": {
"pa_provider": {
"pass_at_1": 0.52,
"n_trials": 25,
"n_tasks": 25,
"mean_cost_usd": 0.0,
"mean_walltime_s": 0.0
},
"pa_um": {
"pass_at_1": 0.52,
"n_trials": 25,
"n_tasks": 25,
"mean_cost_usd": 0.0,
"mean_walltime_s": 0.0
},
"cm": {
"pass_at_1": 0.72,
"n_trials": 25,
"n_tasks": 25,
"mean_cost_usd": 0.0,
"mean_walltime_s": 0.0
}
},
"mean_cost_usd": 0.0,
"mean_walltime_s": 0.0
},
"provenance": {
"chi_bench_git_sha": null,
"image_digest": null,
"judge_model": "claude-opus-4-7",
"harness_version": "0.1.0",
"selection": {
"description": "Integrated results for all three benchmark domains in the current submission directory."
}
}
}
Binary file not shown.
Original file line number Diff line number Diff line change
@@ -0,0 +1,103 @@
{
"id": "29bb92f3-097e-4834-abd2-489db18d1cf0",
"task_name": "actava-ai/cm_afib_moderate_anxious_001",
"trial_name": "cm_afib_moderate_anxious_001__SEG6EVC",
"trial_uri": "trials/cm/cm_afib_moderate_anxious_001__SEG6EVC",
"task_id": {
"path": "data/care_management/tasks/cm_afib_moderate_anxious_001"
},
"source": null,
"task_checksum": "a43c14fadec72f11c1de9af6ee038fd27d3c453ce78dd9de56d454e395e33846",
"config": {
"task": {
"path": "data/care_management/tasks/cm_afib_moderate_anxious_001",
"git_url": null,
"git_commit_id": null,
"name": null,
"ref": null,
"overwrite": false,
"download_dir": null,
"source": null
},
"trial_name": "cm_afib_moderate_anxious_001__SEG6EVC",
"trials_dir": "trials/cm",
"timeout_multiplier": 2.0,
"agent_timeout_multiplier": null,
"verifier_timeout_multiplier": null,
"agent_setup_timeout_multiplier": null,
"environment_build_timeout_multiplier": null,
"agent": {
"name": null,
"import_path": "chi_bench.experiment.agents.hermes_harness:HermesHarness",
"model_name": "openai/qwen3.5_27b_sft_0609",
"override_timeout_sec": null,
"override_setup_timeout_sec": null,
"max_timeout_sec": null,
"kwargs": {},
"env": {}
},
"environment": {
"type": null,
"import_path": "chi_bench.experiment.docker_env:ChiBenchDockerEnvironment",
"force_build": false,
"delete": true,
"override_cpus": null,
"override_memory_mb": null,
"override_storage_mb": null,
"override_gpus": null,
"suppress_override_warnings": false,
"mounts_json": null,
"env": {},
"kwargs": {}
},
"verifier": {
"override_timeout_sec": null,
"max_timeout_sec": null,
"env": {},
"disable": false
},
"artifacts": [],
"job_id": "c1dd117d-2a24-4340-8cac-98389c0ed26d"
},
"agent_info": {
"name": "hermes",
"version": "Hermes Agent v0.17.0 (2026.6.19)\nProject: /usr/local/lib/hermes-agent/venv/lib/python3.12/site-packages\nPython: 3.12.13\nOpenAI SDK: 2.24.0\nUpdate available: 1 commit behind — run 'uv pip install --upgrade hermes-agent'",
"model_info": {
"name": "qwen3.5_27b_sft_0609",
"provider": "openai"
}
},
"agent_result": {
"n_input_tokens": 0,
"n_cache_tokens": null,
"n_output_tokens": 0,
"cost_usd": null,
"rollout_details": null,
"metadata": null
},
"verifier_result": {
"rewards": {
"reward": 0.0
}
},
"exception_info": null,
"started_at": "2026-07-19T05:48:38.555475Z",
"finished_at": "2026-07-19T05:50:18.690223Z",
"environment_setup": {
"started_at": "2026-07-19T05:48:38.625730Z",
"finished_at": "2026-07-19T05:48:48.683974Z"
},
"agent_setup": {
"started_at": "2026-07-19T05:48:48.683988Z",
"finished_at": "2026-07-19T05:48:52.754655Z"
},
"agent_execution": {
"started_at": "2026-07-19T05:48:52.754737Z",
"finished_at": "2026-07-19T05:50:14.897896Z"
},
"verifier": {
"started_at": "2026-07-19T05:50:15.296800Z",
"finished_at": "2026-07-19T05:50:18.353967Z"
},
"step_results": null
}
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
{"reward": 0.0}
Original file line number Diff line number Diff line change
@@ -0,0 +1,132 @@
{
"binary_reward": 0.0,
"fractional_reward": 0.3333333333333333,
"passed_checks": 2,
"total_checks": 6,
"check_scores": {
"cm.assessment.record_exists": 0.0,
"cm.assessment.completed": null,
"cm.assessment.required_sections_present": null,
"cm.care_plan.record_exists": 0.0,
"cm.care_plan.finalized": null,
"cm.care_plan.problem_count": null,
"cm.care_plan.goal_structure": null,
"cm.care_plan.intervention_structure": null,
"cm.care_plan.escalation_conditions_present": null,
"cm.care_plan.follow_up_cadence_present": null,
"cm.chart_review.record_exists": 1.0,
"cm.cross_stage.target_status": 0.0,
"cm.cross_stage.audit_actions": 0.0,
"cm.cross_stage.no_forbidden_mutations": 1.0,
"judge.cm.chart_review.quality": null,
"judge.cm.outreach.quality": null,
"judge.cm.assessment.quality": null,
"judge.cm.care_plan.quality": null,
"judge.cm.stage_coherence": null
},
"checks": {
"cm.assessment.record_exists": false,
"cm.assessment.completed": "not_applicable",
"cm.assessment.required_sections_present": "not_applicable",
"cm.care_plan.record_exists": false,
"cm.care_plan.finalized": "not_applicable",
"cm.care_plan.problem_count": "not_applicable",
"cm.care_plan.goal_structure": "not_applicable",
"cm.care_plan.intervention_structure": "not_applicable",
"cm.care_plan.escalation_conditions_present": "not_applicable",
"cm.care_plan.follow_up_cadence_present": "not_applicable",
"cm.chart_review.record_exists": true,
"cm.cross_stage.target_status": false,
"cm.cross_stage.audit_actions": false,
"cm.cross_stage.no_forbidden_mutations": true,
"judge.cm.chart_review.quality": "not_applicable",
"judge.cm.outreach.quality": "not_applicable",
"judge.cm.assessment.quality": "not_applicable",
"judge.cm.care_plan.quality": "not_applicable",
"judge.cm.stage_coherence": "not_applicable"
},
"failed_checks": [
"cm.assessment.record_exists",
"cm.care_plan.record_exists",
"cm.cross_stage.target_status",
"cm.cross_stage.audit_actions"
],
"not_applicable_checks": [
"cm.assessment.completed",
"cm.assessment.required_sections_present",
"cm.care_plan.finalized",
"cm.care_plan.problem_count",
"cm.care_plan.goal_structure",
"cm.care_plan.intervention_structure",
"cm.care_plan.escalation_conditions_present",
"cm.care_plan.follow_up_cadence_present",
"judge.cm.chart_review.quality",
"judge.cm.outreach.quality",
"judge.cm.assessment.quality",
"judge.cm.care_plan.quality",
"judge.cm.stage_coherence"
],
"stages": {
"cm_assessment": {
"passed": false,
"checks": {
"cm.assessment.record_exists": false,
"cm.assessment.completed": "not_applicable",
"cm.assessment.required_sections_present": "not_applicable"
},
"passed_count": 0,
"total_count": 1,
"not_applicable_count": 2,
"details": {
"case_id": "CM-CASE-CM_AFIB_MODERATE_ANXIOUS_001",
"record_count": 0
}
},
"cm_care_plan": {
"passed": false,
"checks": {
"cm.care_plan.record_exists": false,
"cm.care_plan.finalized": "not_applicable",
"cm.care_plan.problem_count": "not_applicable",
"cm.care_plan.goal_structure": "not_applicable",
"cm.care_plan.intervention_structure": "not_applicable",
"cm.care_plan.escalation_conditions_present": "not_applicable",
"cm.care_plan.follow_up_cadence_present": "not_applicable"
},
"passed_count": 0,
"total_count": 1,
"not_applicable_count": 6,
"details": {
"case_id": "CM-CASE-CM_AFIB_MODERATE_ANXIOUS_001",
"record_count": 0
}
},
"cm_chart_review": {
"passed": true,
"checks": {
"cm.chart_review.record_exists": true
},
"passed_count": 1,
"total_count": 1,
"not_applicable_count": 0,
"details": {
"case_id": "CM-CASE-CM_AFIB_MODERATE_ANXIOUS_001",
"record_count": 1
}
},
"cm_cross_stage": {
"passed": false,
"checks": {
"cm.cross_stage.target_status": false,
"cm.cross_stage.audit_actions": false,
"cm.cross_stage.no_forbidden_mutations": true
},
"passed_count": 1,
"total_count": 3,
"not_applicable_count": 0,
"details": {
"case_id": "CM-CASE-CM_AFIB_MODERATE_ANXIOUS_001"
}
}
}
}
Binary file not shown.
Loading
Loading