Skip to content

Commit a6a6d03

Browse files
committed
Dogfood fixes: a MiMo provider, native loops on any registry provider, safer source fetches
From D2's dogfooding of real workloads: a built-in MiMo provider (OpenAI- and Anthropic-compatible token plan); native loops fall back to the provider registry; a lock around source checkouts, and abbreviated commits found; Daytona rebuilds an image whose earlier build was canceled; the run, results, and viewer output say more of what happened. The missing-key hint no longer suggests --env.
1 parent ea6998e commit a6a6d03

17 files changed

Lines changed: 615 additions & 57 deletions

File tree

Lines changed: 24 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,24 @@
1+
# Xiaomi MiMo's token plan (Singapore): MiMo models (mimo-v2.6-flash,
2+
# mimo-v2.6-pro, mimo-v2.5, mimo-v2.5-pro). Run live on 2026-10-10: the
3+
# OpenAI-compatible API answered Chat Completions and the Responses API, the
4+
# Anthropic-compatible one Messages with an x-api-key, and `/models` lists the
5+
# models for a valid key. A plan key works only on this host (the pay-as-you-go
6+
# api.xiaomimimo.com and the China plan host reject it), and a plan whose
7+
# quota is spent answers 429 "quota exhausted".
8+
name = "mimo"
9+
description = "Xiaomi MiMo token plan: MiMo models"
10+
11+
[openai]
12+
base_url = "https://token-plan-sgp.xiaomimimo.com/v1"
13+
apis = ["chat", "responses"]
14+
15+
[anthropic]
16+
base_url = "https://token-plan-sgp.xiaomimimo.com/anthropic"
17+
18+
[key]
19+
env = ["MIMO_API_KEY"]
20+
21+
[check]
22+
# The model list: needs a valid key, spends nothing.
23+
endpoint = "openai"
24+
path = "/models"

‎crates/benchflow-agent/src/native/endpoint.rs‎

Lines changed: 104 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -4,8 +4,8 @@
44
//! A model is named `<provider>/<model>` (`vercel/deepseek/deepseek-v4-flash-0731`,
55
//! `anthropic/claude-haiku-4-5`). [`EndpointResolver`] turns that name into
66
//! an [`Endpoint`]. [`BuiltinProviders`] knows the providers BenchFlow ships
7-
//! with until the provider registry (`providers/`) replaces it behind the
8-
//! same trait.
7+
//! with, and falls back to the provider registry (`providers/`) for the
8+
//! rest.
99
1010
use std::fmt;
1111

@@ -97,14 +97,12 @@ impl EndpointError {
9797
match self {
9898
EndpointError::NoProvider { .. } | EndpointError::UnknownProvider { .. } => format!(
9999
"name the model as <provider>/<model> with one of: {}",
100-
BUILTIN
101-
.iter()
102-
.map(|p| p.name)
103-
.collect::<Vec<_>>()
104-
.join(", ")
100+
known_providers().join(", ")
105101
),
102+
// Not `--env`: those values reach the agent's sandbox, and a
103+
// native loop's key must stay on this machine.
106104
EndpointError::NoKey { wanted, .. } => {
107-
format!("export {wanted}=<key> (or pass --env {wanted}=<key>), then run again")
105+
format!("export {wanted}=<key> in this shell, then run again")
108106
}
109107
}
110108
}
@@ -186,10 +184,7 @@ impl EndpointResolver for BuiltinProviders {
186184
});
187185
};
188186
let Some(spec) = BUILTIN.iter().find(|p| p.name == provider) else {
189-
return Err(EndpointError::UnknownProvider {
190-
provider: provider.to_string(),
191-
model: model.to_string(),
192-
});
187+
return resolve_from_registry(provider, rest, model, host);
193188
};
194189
let found = spec.keys.iter().find_map(|name| {
195190
host.get(name)
@@ -238,6 +233,65 @@ impl EndpointResolver for BuiltinProviders {
238233
}
239234
}
240235

236+
/// A provider the native loops have no entry for, from the provider
237+
/// registry (`zai/`, `deepseek/`, `mimo/`): its OpenAI-compatible endpoint
238+
/// when it takes a bearer key, else its Anthropic-compatible one when it takes
239+
/// `x-api-key`, the way the native clients send keys.
240+
fn resolve_from_registry(
241+
provider: &str,
242+
rest: &str,
243+
model: &str,
244+
host: &HostEnv,
245+
) -> Result<Endpoint, EndpointError> {
246+
use crate::providers::{Auth, ProviderRegistry};
247+
let unknown = || EndpointError::UnknownProvider {
248+
provider: provider.to_string(),
249+
model: model.to_string(),
250+
};
251+
let registry = ProviderRegistry::builtin();
252+
let p = registry.find(provider).ok_or_else(unknown)?;
253+
let (api, base_url) = if let Some(e) = p.openai.as_ref().filter(|e| e.auth == Auth::Bearer) {
254+
(Api::OpenAiChat, e.base_url.clone())
255+
} else if let Some(e) = p.anthropic.as_ref().filter(|e| e.auth == Auth::XApiKey) {
256+
// The registry's Anthropic base URL stops before the version.
257+
(Api::AnthropicMessages, format!("{}/v1", e.base_url))
258+
} else {
259+
return Err(unknown());
260+
};
261+
let key = crate::providers::find_provider_key(p, host)
262+
.ok()
263+
.filter(|k| !k.value().trim().is_empty())
264+
.ok_or_else(|| EndpointError::NoKey {
265+
provider: provider.to_string(),
266+
wanted: p.key.env.join(" or "),
267+
})?;
268+
let base_url = host
269+
.get(BASE_URL_OVERRIDE_ENV)
270+
.map(|(v, _)| v.trim().to_string())
271+
.filter(|v| !v.is_empty())
272+
.unwrap_or(base_url);
273+
Ok(Endpoint {
274+
api,
275+
base_url: base_url.trim_end_matches('/').to_string(),
276+
key: ApiKey::new(key.value()),
277+
model: rest.to_string(),
278+
provider: provider.to_string(),
279+
key_label: key.source.clone(),
280+
upstreams: None,
281+
})
282+
}
283+
284+
/// Every provider a native loop can name: its own, then the registry's.
285+
fn known_providers() -> Vec<String> {
286+
let mut names: Vec<String> = BUILTIN.iter().map(|p| p.name.to_string()).collect();
287+
for p in crate::providers::ProviderRegistry::builtin().providers() {
288+
if !names.contains(&p.name) {
289+
names.push(p.name.clone());
290+
}
291+
}
292+
names
293+
}
294+
241295
#[cfg(test)]
242296
mod tests {
243297
use std::collections::BTreeMap;
@@ -332,6 +386,44 @@ mod tests {
332386
assert!(matches!(err, EndpointError::NoProvider { .. }));
333387
}
334388

389+
#[test]
390+
fn registry_providers_work_for_native_loops_too() {
391+
// MiMo's token plan: OpenAI-compatible with a bearer key.
392+
let e = BuiltinProviders
393+
.resolve(
394+
"mimo/mimo-v2.6-flash",
395+
&host(&[("MIMO_API_KEY", "k-123456789")]),
396+
)
397+
.unwrap();
398+
assert_eq!(e.api, Api::OpenAiChat);
399+
assert_eq!(
400+
e.url(),
401+
"https://token-plan-sgp.xiaomimimo.com/v1/chat/completions"
402+
);
403+
assert_eq!(e.model, "mimo-v2.6-flash");
404+
assert_eq!(e.key_label, "MIMO_API_KEY (environment)");
405+
// Z.ai and DeepSeek, which were refused before.
406+
for (model, var) in [
407+
("zai/glm-5.3", "ZAI_API_KEY"),
408+
("deepseek/deepseek-chat", "DEEPSEEK_API_KEY"),
409+
] {
410+
let e = BuiltinProviders
411+
.resolve(model, &host(&[(var, "k-123456789")]))
412+
.unwrap();
413+
assert_eq!(e.api, Api::OpenAiChat, "{model}");
414+
}
415+
let err = BuiltinProviders
416+
.resolve("mimo/mimo-v2.6-flash", &host(&[]))
417+
.unwrap_err();
418+
assert!(
419+
err.next_step().contains("MIMO_API_KEY"),
420+
"{}",
421+
err.next_step()
422+
);
423+
let err = BuiltinProviders.resolve("nope/x", &host(&[])).unwrap_err();
424+
assert!(err.next_step().contains("mimo"), "{}", err.next_step());
425+
}
426+
335427
#[test]
336428
fn the_override_points_any_provider_at_another_server() {
337429
let e = BuiltinProviders

‎crates/benchflow-agent/src/providers.rs‎

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -2,7 +2,7 @@
22
//!
33
//! A provider is a TOML file that says where its endpoints are, where its key
44
//! comes from, and how to check the key. The built-in providers (`vercel`,
5-
//! `openai`, `anthropic`, `openrouter`, `zai`, `deepseek`) are embedded in
5+
//! `openai`, `anthropic`, `openrouter`, `zai`, `deepseek`, `mimo`) are embedded in
66
//! the binary from `crates/benchflow-agent/providers/`; a project adds or
77
//! replaces providers with its own files ([`ProviderRegistry::with_files`]).
88
//!
@@ -364,6 +364,7 @@ const BUILTINS: &[(&str, &str)] = &[
364364
),
365365
("zai.toml", include_str!("../providers/zai.toml")),
366366
("deepseek.toml", include_str!("../providers/deepseek.toml")),
367+
("mimo.toml", include_str!("../providers/mimo.toml")),
367368
];
368369

369370
/// Every provider a run knows: the built-ins, then the project's.

‎crates/benchflow-agent/src/providers/tests.rs‎

Lines changed: 17 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -44,7 +44,7 @@ fn host(vars: &[(&str, &str)], home: Option<PathBuf>) -> HostEnv {
4444
}
4545

4646
#[test]
47-
fn builtins_are_the_six_providers_with_their_endpoints() {
47+
fn builtins_are_the_seven_providers_with_their_endpoints() {
4848
let names: Vec<String> = Provider::builtins().into_iter().map(|p| p.name).collect();
4949
assert_eq!(
5050
names,
@@ -54,7 +54,8 @@ fn builtins_are_the_six_providers_with_their_endpoints() {
5454
"anthropic",
5555
"openrouter",
5656
"zai",
57-
"deepseek"
57+
"deepseek",
58+
"mimo"
5859
]
5960
);
6061
let registry = providers();
@@ -115,6 +116,18 @@ fn builtins_are_the_six_providers_with_their_endpoints() {
115116
.apis,
116117
[OpenAiApi::Chat]
117118
);
119+
// MiMo's token plan serves both OpenAI APIs and Anthropic's, on one host.
120+
let mimo = registry.find("mimo").unwrap();
121+
assert_eq!(
122+
mimo.openai.as_ref().unwrap().base_url,
123+
"https://token-plan-sgp.xiaomimimo.com/v1"
124+
);
125+
assert_eq!(
126+
mimo.openai.as_ref().unwrap().apis,
127+
[OpenAiApi::Chat, OpenAiApi::Responses]
128+
);
129+
assert_eq!(mimo.anthropic.as_ref().unwrap().auth, Auth::XApiKey);
130+
assert_eq!(mimo.key.env, ["MIMO_API_KEY"]);
118131
assert_eq!(
119132
ManifestSource::BuiltinProvider("vercel.toml").to_string(),
120133
"built-in providers/vercel.toml"
@@ -191,7 +204,7 @@ fn a_project_provider_replaces_a_builtin_of_the_same_name() {
191204
)
192205
.unwrap();
193206
let registry = ProviderRegistry::with_files([dir.path()]).unwrap();
194-
assert_eq!(registry.providers().len(), 7);
207+
assert_eq!(registry.providers().len(), 8);
195208
let vercel = registry.find("vercel").unwrap();
196209
assert_eq!(vercel.key.env, ["MY_GATEWAY_KEY"]);
197210
assert!(vercel.anthropic.is_none());
@@ -365,7 +378,7 @@ fn an_agent_that_cannot_use_the_provider_is_refused_with_its_options() {
365378
);
366379
assert_eq!(
367380
error.next_step(),
368-
"pick another agent, or a provider `codex` can use: vercel, openai, openrouter"
381+
"pick another agent, or a provider `codex` can use: vercel, openai, openrouter, mimo"
369382
);
370383

371384
let error = route(&agent("claude"), Some("vercel/"), &providers()).unwrap_err();

‎crates/benchflow-cli/src/cli.rs‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -438,7 +438,7 @@ pub struct RunArgs {
438438
pub agent: Option<String>,
439439

440440
/// The model, in the agent's own naming, or `<provider>/<model>` to run on a
441-
/// provider's key (vercel, openai, anthropic, openrouter, zai, deepseek)
441+
/// provider's key (vercel, openai, anthropic, openrouter, zai, deepseek, mimo)
442442
#[arg(short, long)]
443443
pub model: Option<String>,
444444

‎crates/benchflow-cli/src/cmd/results.rs‎

Lines changed: 27 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -392,11 +392,10 @@ fn compare(views: &[JobView]) -> Result<Exit> {
392392
Some(p) => {
393393
let parts = [
394394
format!(
395-
" A − {b}: {:+.3} mean reward over {} (95% CI {:+.3} to {:+.3})",
395+
" A − {b}: {:+.3} mean reward over {} ({})",
396396
p.diff,
397397
plural(p.tasks, "shared task"),
398-
p.low,
399-
p.high
398+
paired_interval(p.tasks, p.low, p.high)
400399
),
401400
format!(
402401
"A better on {}, {b} better on {}, tied on {}",
@@ -487,3 +486,28 @@ fn job_json(v: &JobView) -> serde_json::Value {
487486
"cost_recorded": v.records.iter().filter(|r| r.usage.cost_usd.is_some()).count(),
488487
})
489488
}
489+
490+
/// The paired difference's interval, or why there is none: one shared task
491+
/// has no spread to make one from (the bounds would both be the difference).
492+
fn paired_interval(tasks: usize, low: f64, high: f64) -> String {
493+
if tasks < 2 {
494+
"no interval from one task".to_string()
495+
} else {
496+
format!("95% CI {low:+.3} to {high:+.3}")
497+
}
498+
}
499+
500+
#[cfg(test)]
501+
mod paired_tests {
502+
#[test]
503+
fn one_shared_task_gives_no_interval() {
504+
assert_eq!(
505+
super::paired_interval(1, 0.0, 0.0),
506+
"no interval from one task"
507+
);
508+
assert_eq!(
509+
super::paired_interval(3, -0.5, 0.25),
510+
"95% CI -0.500 to +0.250"
511+
);
512+
}
513+
}

‎crates/benchflow-cli/src/cmd/run.rs‎

Lines changed: 69 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -188,12 +188,17 @@ pub async fn run(args: RunArgs) -> Result<Exit> {
188188
}
189189
let (tasks, bad) = context::load_tasks(&roots)?;
190190
for r in &bad {
191-
let errs = r.diagnostics.iter().filter(|d| d.is_error()).count();
191+
let errs: Vec<(&str, &str)> = r
192+
.diagnostics
193+
.iter()
194+
.filter(|d| d.is_error())
195+
.map(|d| (d.code, d.message.as_str()))
196+
.collect();
192197
note!(
193198
" {} skipping {}: {} ({})",
194199
warn_mark(),
195200
ui::path(&r.root),
196-
plural(errs, "error"),
201+
skip_reason(&errs),
197202
dim(format!("bench check {}", ui::path(&r.root)))
198203
);
199204
}
@@ -414,10 +419,11 @@ pub async fn run(args: RunArgs) -> Result<Exit> {
414419
let what = format!(
415420
"{} {} on {}",
416421
bold("Running"),
417-
match &model {
418-
Some(m) => format!("{} ({m})", attempt.label()),
419-
None => attempt.label(),
420-
},
422+
attempt_title(
423+
&attempt.label(),
424+
model.as_deref(),
425+
attempt.model().as_deref()
426+
),
421427
plural(tasks.len() * trials, "trial"),
422428
);
423429
let place = format!("{} · {} at a time", provider.name(), st.concurrency);
@@ -1122,3 +1128,60 @@ mod exit_tests {
11221128
assert!(run_exit(&[Status::Scored], 1) == Exit::Failed);
11231129
}
11241130
}
1131+
1132+
/// Why a task was skipped, on its one line: the first error itself, so a
1133+
/// dataset whose tasks all fail the same way says how without a `bench
1134+
/// check` per task.
1135+
fn skip_reason(errors: &[(&str, &str)]) -> String {
1136+
match errors {
1137+
[] => "it did not load".to_string(),
1138+
[(code, message)] => format!("[{code}] {message}"),
1139+
[(code, message), rest @ ..] => {
1140+
let more = if rest.len() == 1 { "error" } else { "errors" };
1141+
format!("[{code}] {message} (and {} more {more})", rest.len())
1142+
}
1143+
}
1144+
}
1145+
1146+
/// What runs, for the `Running ...` line: the agent and its model. With no
1147+
/// `-m`, the agent's own default model (Claude Code's is Haiku), which
1148+
/// nothing else on screen names.
1149+
fn attempt_title(label: &str, asked: Option<&str>, running: Option<&str>) -> String {
1150+
match (asked, running) {
1151+
(Some(m), _) => format!("{label} ({m})"),
1152+
(None, Some(m)) => format!("{label} ({m}, its default)"),
1153+
(None, None) => label.to_string(),
1154+
}
1155+
}
1156+
1157+
#[cfg(test)]
1158+
mod title_tests {
1159+
use super::attempt_title;
1160+
1161+
#[test]
1162+
fn a_skipped_task_says_its_first_error() {
1163+
let renamed = ("X203", "the `environment` key was renamed to `sandbox`");
1164+
assert_eq!(
1165+
super::skip_reason(&[renamed]),
1166+
"[X203] the `environment` key was renamed to `sandbox`"
1167+
);
1168+
assert_eq!(
1169+
super::skip_reason(&[renamed, ("X201", "x"), ("X201", "y")]),
1170+
"[X203] the `environment` key was renamed to `sandbox` (and 2 more errors)"
1171+
);
1172+
assert_eq!(super::skip_reason(&[]), "it did not load");
1173+
}
1174+
1175+
#[test]
1176+
fn the_running_line_names_a_default_model() {
1177+
assert_eq!(
1178+
attempt_title("builtin", Some("mimo/x"), Some("mimo/x")),
1179+
"builtin (mimo/x)"
1180+
);
1181+
assert_eq!(
1182+
attempt_title("claude", None, Some("claude-haiku-4-5-20251001")),
1183+
"claude (claude-haiku-4-5-20251001, its default)"
1184+
);
1185+
assert_eq!(attempt_title("oracle", None, None), "oracle");
1186+
}
1187+
}

0 commit comments

Comments
 (0)