diff --git a/CHANGELOG.md b/CHANGELOG.md index 6b21a15..f883e23 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,6 +1,20 @@ # Changelog -## 0.9.0 — Unreleased +## 0.9.1 — Unreleased + +- Accept first-party Claude Team subscriptions alongside Pro and Max while + preserving exact first-party authentication checks and account-data redaction. +- Restore macOS Keychain discovery with canonical POSIX `USER` and `LOGNAME` + values without inheriting ambient identity, credential, provider, proxy, or + routing overrides. +- Accept either reviewed Fable runtime primary (`claude-fable-5` or + `claude-opus-4-8`) plus only the exact reviewed Haiku helper, while keeping + the separate Opus route pinned to `claude-opus-5` with no helper. +- Require schema-validated Advisor decisions, normalize one unambiguous Claude + result object/event, and reject malformed, conflicting, or raw-prose review + output without exposing model-authored content in errors. + +## 0.9.0 — 2026-07-25 - Add Claude Opus 5 as a sealed first-party subscription Planner or Advisor through Claude Code 2.1.219 or newer, with exact effort validation and diff --git a/README.md b/README.md index 4498b47..7ab15e0 100644 --- a/README.md +++ b/README.md @@ -200,10 +200,12 @@ Fable 5 is the bundled cross-provider exception retained for compatibility; Opus 5 is the second sealed bundled exception added in version 0.9.0. The bundled Claude bridge starts each authentication and model subprocess with -only a minimal platform environment. It preserves canonical `HOME` on POSIX or -`USERPROFILE` on Windows so the official CLI can find the user's first-party login, -but it does not inherit credential, config-redirection, provider/model/effort, -endpoint/gateway, proxy/CA/mTLS, or telemetry override families. +only a minimal platform environment. It preserves `HOME` plus canonical +operating-system `USER` and `LOGNAME` on POSIX, or `USERPROFILE` on Windows, so +the official CLI can find the user's first-party login. It does not trust ambient +POSIX identity values or inherit credential, config-redirection, +provider/model/effort, endpoint/gateway, proxy/CA/mTLS, or telemetry override +families. Models already available through Codex can still become ordinary user-owned roles: diff --git a/plugins/codex-orchestration/.codex-plugin/plugin.json b/plugins/codex-orchestration/.codex-plugin/plugin.json index 4625dad..464ee7b 100644 --- a/plugins/codex-orchestration/.codex-plugin/plugin.json +++ b/plugins/codex-orchestration/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "codex-orchestration", - "version": "0.9.0", + "version": "0.9.1", "description": "Give Codex and audited external models safe, provider-pinned roles.", "author": { "name": "CJ Zafir", diff --git a/plugins/codex-orchestration/skills/codex-orchestration/SKILL.md b/plugins/codex-orchestration/skills/codex-orchestration/SKILL.md index 1d06db4..e3f059a 100644 --- a/plugins/codex-orchestration/skills/codex-orchestration/SKILL.md +++ b/plugins/codex-orchestration/skills/codex-orchestration/SKILL.md @@ -355,7 +355,7 @@ python3 /scripts/configure_native_routing.py \ --apply ``` -Add `--advisor-model` and `--advisor-effort` for a same-provider Codex advisor. For Claude Fable 5, use `--advisor-fable`; add `--advisor-effort low|medium|high|xhigh|max` when the user chooses one. Omitting Fable effort defaults to `high`, while user-facing `ultra` is normalized to Claude Code's `max`. The configurator verifies that the installed Claude Code CLI advertises the selected effective effort. It also requires Claude Code to be logged in through a first-party Pro or Max account, chooses an available Python 3.11+ MCP launcher, and performs only an auth/capability check during setup. It never extracts a token, writes a credential, or makes a model call during setup or status. Omission persists `advisor: none`. +Add `--advisor-model` and `--advisor-effort` for a same-provider Codex advisor. For Claude Fable 5, use `--advisor-fable`; add `--advisor-effort low|medium|high|xhigh|max` when the user chooses one. Omitting Fable effort defaults to `high`, while user-facing `ultra` is normalized to Claude Code's `max`. The configurator verifies that the installed Claude Code CLI advertises the selected effective effort. It also requires Claude Code to be logged in through a first-party Pro, Max, or Team account, chooses an available Python 3.11+ MCP launcher, and performs only an auth/capability check during setup. It never extracts a token, writes a credential, or makes a model call during setup or status. Omission persists `advisor: none`. For Claude Opus 5 Advisor, use `--advisor-opus` with an optional exact `--advisor-effort low|medium|high|xhigh|max`. The default is `high`. Setup also @@ -466,10 +466,10 @@ Report authentication as `first-party login ready`; do not expose or restate Cla Prerequisites: - the official `claude` CLI is installed; -- `claude auth status` reports a first-party Pro or Max login; +- `claude auth status --json` reports a first-party Pro, Max, or Team login; - a Python 3.11+ launcher is available. -The plugin packages three disabled MCP launcher variants for macOS, Linux, and Windows. Setup enables exactly one compatible variant through the plugin's namespaced config when either bundled Claude model is selected. At planning or review time the MCP server gives authentication and model subprocesses only a minimal platform environment. It preserves canonical `HOME` on POSIX or `USERPROFILE` on Windows for first-party login discovery, but does not inherit credential, config-redirection, provider/model/effort, endpoint/gateway, proxy/CA/mTLS, or telemetry override families. It invokes `claude --print --model ` with `--safe-mode`, no tools, no session persistence, prompt suggestions disabled, and JSON output. Each saved seat pins its model and effort; the root cannot replace them through tool arguments. +The plugin packages three disabled MCP launcher variants for macOS, Linux, and Windows. Setup enables exactly one compatible variant through the plugin's namespaced config when either bundled Claude model is selected. At planning or review time the MCP server gives authentication and model subprocesses only a minimal platform environment. It preserves `HOME` plus canonical operating-system `USER` and `LOGNAME` on POSIX, or `USERPROFILE` on Windows, for first-party login discovery. It does not trust ambient POSIX identity values and does not inherit credential, config-redirection, provider/model/effort, endpoint/gateway, proxy/CA/mTLS, or telemetry override families. It invokes `claude --print --model ` with `--safe-mode`, no tools, no session persistence, prompt suggestions disabled, and JSON output. Advisor review additionally requires Claude Code's `--json-schema` capability. Each saved seat pins its model and effort; the root cannot replace them through tool arguments. Fable effort is configurable per setup. The default is `high`; supported Claude Code values are `low`, `medium`, `high`, `xhigh`, and `max`. Accept `ultra` as an alias for `max`, save the effective Claude Code value, and disclose the alias mapping in setup output. Existing saved `max` routes remain valid. @@ -478,10 +478,12 @@ exactly `low`, `medium`, `high`, `xhigh`, and `max`; no alias is accepted. Setup requires only that the selected sealed effort appear in the installed CLI's advertised set. Extra advertised values do not expand the sealed set. -The bridge exposes only bounded, read-only planning operations. `create_plan` accepts one self-contained packet and requires `PLAN_DRAFT`. `revise_plan` requires the task, canonical current plan, latest critique, and compact findings history, then requires `PLAN_REVISION` plus a findings ledger and revised plan. `review_plan` remains the Advisor operation and requires `PLAN_APPROVED` or `PLAN_REVISE`. Every call uses the same full saved-state validator as native status/repair/disable, then requires runtime `modelUsage` to contain the pinned primary plus only that model's explicit exact helper allowlist. Fable permits its independently observed `claude-haiku-4-5-20251001` helper. No Opus helper identity is independently established, so Opus currently permits only `claude-opus-5` and fails closed if any additional runtime model appears. Return every observed ID in `used_models`; an unknown additional or missing primary model makes the seat unavailable. Any auth, transport, state, format, or model-confirmation failure makes that seat unavailable; it never counts as approval. The bridge returns no account identifier or credential. Local mocked verification does not prove a positive live Opus invocation. +The bridge exposes only bounded, read-only planning operations. `create_plan` accepts one self-contained packet and requires `PLAN_DRAFT`. `revise_plan` requires the task, canonical current plan, latest critique, and compact findings history, then requires `PLAN_REVISION` plus a findings ledger and revised plan. `review_plan` remains the Advisor operation and requires a locally revalidated JSON Schema object containing exactly `PLAN_APPROVED` or `PLAN_REVISE` plus a non-empty body; raw prose never counts as a decision. Every call uses the same full saved-state validator as native status/repair/disable, then requires runtime `modelUsage` to contain a reviewed Fable primary identity (`claude-fable-5` or `claude-opus-4-8`) or the exact Opus primary, plus only that model's explicit exact helper allowlist. Fable permits its independently observed `claude-haiku-4-5-20251001` helper. No Opus helper identity is independently established, so Opus currently permits only `claude-opus-5` and fails closed if any additional runtime model appears. Return every observed ID in `used_models`; an unknown additional or missing primary model makes the seat unavailable. Any auth, transport, state, format, or model-confirmation failure makes that seat unavailable; it never counts as approval. The bridge returns no account identifier or credential. Local mocked verification does not prove a positive live Opus invocation. -The legacy Fable contract requires runtime `modelUsage` to contain the pinned `claude-fable-5` -primary; Opus applies the same primary-presence rule to `claude-opus-5`. +The configured Fable route remains `claude-fable-5`, while runtime `modelUsage` +may confirm either reviewed Fable primary identity. This does not make +`claude-opus-4-8` an Opus route alias. Opus still requires the exact +`claude-opus-5` primary. Plugin and policy updates cannot replace the MCP process already loaded into the current task. If a bundled Claude call fails after an update or repair, run fresh diff --git a/plugins/codex-orchestration/skills/codex-orchestration/references/providers-and-models.md b/plugins/codex-orchestration/skills/codex-orchestration/references/providers-and-models.md index e9da6d7..b3884f8 100644 --- a/plugins/codex-orchestration/skills/codex-orchestration/references/providers-and-models.md +++ b/plugins/codex-orchestration/skills/codex-orchestration/references/providers-and-models.md @@ -228,22 +228,25 @@ On Windows, in-place update and removal stage the replacement beside the existin Direct v2 `model` overrides retain the parent's provider. They are the simplest route for an OpenAI root and OpenAI Luna/Terra child. -Claude Fable 5 and Claude Opus 5 are explicit built-in exceptions for Planner or Advisor. The plugin does not pretend either is a Codex model or translate Anthropic into the Responses protocol. Instead, a disabled-by-default local MCP server invokes the official `claude` CLI with the user's first-party Pro or Max login. Setup enables one Python 3.11+ launcher variant, and disable restores every prior plugin override value. The historical `fable-advisor-*` IDs are compatibility names shared by both sealed models. Codex's TOML editor can retain an inert empty table header after its final key is deleted; the configurator does not risk a broad TOML rewrite for cosmetic cleanup. - -The bridge starts both `claude auth status` and model calls with only a minimal -platform environment. It preserves canonical `HOME` on POSIX or `USERPROFILE` on -Windows so the official CLI can discover the first-party login, but it does not -inherit credential, config-redirection, provider/model/effort, endpoint/gateway, -proxy/CA/mTLS, or telemetry override families. It pins the saved primary and effort, -disables tools and session persistence, disables prompt suggestions, and requires -JSON runtime metadata to contain that primary. Claude Code currently reports the -internal helper `claude-haiku-4-5-20251001` during valid Fable calls; the bridge -permits only that exact helper ID. No Opus helper identity is independently -established, so the Opus runtime allowlist currently contains only `claude-opus-5`. -Any missing primary or unknown additional model fails closed. Helper rotation -therefore requires a reviewed plugin update rather than a wildcard. Setup and status -never make a model call, and mocked local tests do not constitute a positive live -Opus invocation. +Claude Fable 5 and Claude Opus 5 are explicit built-in exceptions for Planner or Advisor. The plugin does not pretend either is a Codex model or translate Anthropic into the Responses protocol. Instead, a disabled-by-default local MCP server invokes the official `claude` CLI with the user's first-party Pro, Max, or Team login. Setup enables one Python 3.11+ launcher variant, and disable restores every prior plugin override value. The historical `fable-advisor-*` IDs are compatibility names shared by both sealed models. Codex's TOML editor can retain an inert empty table header after its final key is deleted; the configurator does not risk a broad TOML rewrite for cosmetic cleanup. + +The bridge starts both `claude auth status --json` and model calls with only a +minimal platform environment. It preserves `HOME` plus canonical +operating-system `USER` and `LOGNAME` on POSIX, or `USERPROFILE` on Windows, so +the official CLI can discover the first-party login. Ambient POSIX identity, +credential, config-redirection, provider/model/effort, endpoint/gateway, +proxy/CA/mTLS, and telemetry override families are not trusted. It pins the +saved route and effort, disables tools and session persistence, disables prompt +suggestions, and requires JSON runtime metadata to contain an allowed primary. +For the Fable route, the reviewed primary identities are `claude-fable-5` and +its resolved runtime identity `claude-opus-4-8`; only the exact internal helper +`claude-haiku-4-5-20251001` is additionally permitted. The separate Opus route +still requires `claude-opus-5`, with no helper. Advisor decisions use +`--json-schema` and are locally revalidated; raw prose is not approval. Any +missing primary or unknown additional model fails closed. Identity rotation +therefore requires a reviewed plugin update rather than a wildcard. Setup and +status never make a model call, and mocked local tests do not constitute a +positive live Opus invocation. An MCP process is loaded for the lifetime of its Codex task. Updating the plugin or repairing policy state cannot replace that already loaded process. If a current-task diff --git a/plugins/codex-orchestration/skills/codex-orchestration/scripts/configure_native_routing.py b/plugins/codex-orchestration/skills/codex-orchestration/scripts/configure_native_routing.py index c9b4e95..36b77e6 100644 --- a/plugins/codex-orchestration/skills/codex-orchestration/scripts/configure_native_routing.py +++ b/plugins/codex-orchestration/skills/codex-orchestration/scripts/configure_native_routing.py @@ -456,7 +456,7 @@ def __init__( "clientInfo": { "name": "codex_orchestration_installer", "title": "Codex Orchestration Installer", - "version": "0.9.0", + "version": "0.9.1", }, "capabilities": {"experimentalApi": True}, }, @@ -1005,6 +1005,7 @@ def verify_claude_prerequisites(model: str, effort: str) -> dict[str, str]: "--no-session-persistence", "--prompt-suggestions", "--output-format", + "--json-schema", "--system-prompt", ) advertised_options = set( diff --git a/plugins/codex-orchestration/skills/codex-orchestration/scripts/external_subscription.py b/plugins/codex-orchestration/skills/codex-orchestration/scripts/external_subscription.py index 182bcdc..e4252e1 100644 --- a/plugins/codex-orchestration/skills/codex-orchestration/scripts/external_subscription.py +++ b/plugins/codex-orchestration/skills/codex-orchestration/scripts/external_subscription.py @@ -142,13 +142,23 @@ def invoke( result.get("effort") == selected_effort, "subscription runtime effort drifted", ) + used_models = result.get("used_models") _require( - model in result.get("used_models", []), + type(used_models) is list + and bool(used_models) + and all(type(value) is str and bool(value.strip()) for value in used_models), + "subscription runtime metadata is invalid", + ) + reviewed_primaries = ( + fable_advisor_mcp.REVIEWED_PRIMARY_MODELS_BY_ROUTE[model] + ) + _require( + bool(set(used_models).intersection(reviewed_primaries)), "subscription runtime metadata omitted the primary model", ) allowed_runtime = fable_advisor_mcp.ALLOWED_RUNTIME_MODELS_BY_PRIMARY[model] _require( - set(result.get("used_models", [])).issubset(allowed_runtime), + set(used_models).issubset(allowed_runtime), "subscription runtime metadata included an unsealed helper model", ) return result diff --git a/plugins/codex-orchestration/skills/codex-orchestration/scripts/fable_advisor_mcp.py b/plugins/codex-orchestration/skills/codex-orchestration/scripts/fable_advisor_mcp.py index 1b268b2..14b6d04 100644 --- a/plugins/codex-orchestration/skills/codex-orchestration/scripts/fable_advisor_mcp.py +++ b/plugins/codex-orchestration/skills/codex-orchestration/scripts/fable_advisor_mcp.py @@ -33,7 +33,16 @@ # some calls. Keep the runtime policy explicit and fail closed if that identity # rotates or any other model appears. FABLE_HELPER_MODEL = "claude-haiku-4-5-20251001" -ALLOWED_RUNTIME_MODELS = frozenset({FABLE_MODEL, FABLE_HELPER_MODEL}) +FABLE_RESOLVED_PRIMARY_MODEL = "claude-opus-4-8" +REVIEWED_PRIMARY_MODELS_BY_ROUTE = { + FABLE_MODEL: frozenset({FABLE_MODEL, FABLE_RESOLVED_PRIMARY_MODEL}), + # The resolved Fable identity is not an alias for the separately sealed + # Opus route. Opus remains primary-only until independently re-qualified. + OPUS_MODEL: frozenset({OPUS_MODEL}), +} +ALLOWED_RUNTIME_MODELS = frozenset( + {*REVIEWED_PRIMARY_MODELS_BY_ROUTE[FABLE_MODEL], FABLE_HELPER_MODEL} +) ALLOWED_RUNTIME_MODELS_BY_PRIMARY = { FABLE_MODEL: ALLOWED_RUNTIME_MODELS, # No Opus helper identity has been independently verified. Fail closed if @@ -44,6 +53,18 @@ AUTH_TIMEOUT_SECONDS = 20 # Applies to the combined user-controlled text sent by one model operation. MAX_INPUT_CHARS = 200_000 +PLAN_REVIEW_SCHEMA = { + "type": "object", + "properties": { + "signal": { + "type": "string", + "enum": ["PLAN_APPROVED", "PLAN_REVISE"], + }, + "body": {"type": "string", "minLength": 1}, + }, + "required": ["signal", "body"], + "additionalProperties": False, +} SENSITIVE_ENV = { "ANTHROPIC_API_KEY", "ANTHROPIC_AUTH_TOKEN", @@ -105,8 +126,7 @@ ADVISOR_SYSTEM_PROMPT = """You are the configured Claude model acting only as a plan advisor to Codex's root orchestrator. Review the supplied self-contained packet for material correctness, missing constraints, unsafe sequencing, ownership conflicts, and verification gaps. Do not edit files, call tools, spawn agents, contact the Planner or executors, or attempt implementation. -Your first non-empty line must be exactly PLAN_APPROVED or PLAN_REVISE. -Use PLAN_APPROVED only when no material gap is present. Use PLAN_REVISE when correction is needed. For PLAN_REVISE, assign every material finding a stable, unique finding ID and give a concrete correction. On later rounds, preserve IDs from the supplied cumulative ledger. Ignore style preferences. Report only to the root orchestrator.""" +Return the required structured fields `signal` and `body`. Use signal PLAN_APPROVED only when no material gap is present. Use PLAN_REVISE when correction is needed. The body must be non-empty. For PLAN_REVISE, assign every material finding a stable, unique finding ID and give a concrete correction. On later rounds, preserve IDs from the supplied cumulative ledger. Ignore style preferences. Report only to the root orchestrator.""" PLANNER_CREATE_SYSTEM_PROMPT = """You are the configured Claude model acting only as a plan author for Codex's root orchestrator. Create a concrete implementation plan from the supplied self-contained packet. Include constraints, ownership, sequencing, acceptance criteria, security and compatibility boundaries, and behavioral plus regression verification. Do not edit files, call tools, spawn agents, contact the Advisor or executors, or attempt implementation. @@ -142,6 +162,22 @@ def codex_home() -> Path: return Path(value).expanduser() if value else Path.home() / ".codex" +def _canonical_posix_identity() -> str: + try: + import pwd + + name = pwd.getpwuid(os.getuid()).pw_name + except (ImportError, KeyError, OSError, AttributeError) as exc: + raise AdvisorError( + "Could not determine the canonical POSIX login identity for Claude Code." + ) from exc + if not isinstance(name, str) or not name.strip(): + raise AdvisorError( + "Could not determine the canonical POSIX login identity for Claude Code." + ) + return name + + def sanitized_environment() -> dict[str, str]: common_names = ("PATH", "LANG", "LC_ALL", "LC_CTYPE") if os.name == "nt": @@ -166,6 +202,9 @@ def sanitized_environment() -> dict[str, str]: for name in (*common_names, "HOME", "TMPDIR") if name in os.environ } + identity = _canonical_posix_identity() + env["USER"] = identity + env["LOGNAME"] = identity env["CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC"] = "1" return env @@ -215,17 +254,21 @@ def _run_json(command: list[str], *, timeout: int) -> dict[str, Any]: def check_claude_auth(claude: Path | None = None) -> dict[str, str]: executable = claude or resolve_claude() - payload = _run_json([str(executable), "auth", "status"], timeout=AUTH_TIMEOUT_SECONDS) + payload = _run_json( + [str(executable), "auth", "status", "--json"], + timeout=AUTH_TIMEOUT_SECONDS, + ) subscription = payload.get("subscriptionType") if not ( payload.get("loggedIn") is True and payload.get("authMethod") == "claude.ai" and payload.get("apiProvider") == "firstParty" - and subscription in {"pro", "max"} + and isinstance(subscription, str) + and subscription in {"pro", "max", "team"} ): raise AdvisorError( - "Claude Code must be logged in through a first-party Pro or Max account; " - "run `claude auth login` and try again." + "Claude Code must be logged in through a first-party Pro, Max, or Team " + "account; run `claude auth login` and try again." ) return {"auth_method": "claude.ai", "api_provider": "firstParty"} @@ -315,7 +358,8 @@ def _validate_runtime_models( usage: Any, primary_model: str = FABLE_MODEL ) -> list[str]: allowed_models = ALLOWED_RUNTIME_MODELS_BY_PRIMARY.get(primary_model) - if allowed_models is None: + reviewed_primaries = REVIEWED_PRIMARY_MODELS_BY_ROUTE.get(primary_model) + if allowed_models is None or reviewed_primaries is None: raise AdvisorError("The configured Claude primary model is not sealed.") policy_label = "Fable" if primary_model == FABLE_MODEL else "Claude" primary_label = ( @@ -349,9 +393,10 @@ def _validate_runtime_models( "Runtime metadata has a malformed modelUsage value." ) used_models = sorted(raw_models) - if primary_model not in used_models: + if not set(used_models).intersection(reviewed_primaries): raise AdvisorError( - f"Runtime metadata did not confirm the pinned {primary_label} primary model." + f"Runtime metadata did not confirm the pinned {primary_label} primary " + "model or a reviewed resolved identity." ) if not set(used_models).issubset(allowed_models): raise AdvisorError( @@ -361,6 +406,104 @@ def _validate_runtime_models( return used_models +def _normalize_model_payload( + payload: Any, *, display_name: str, operation: str +) -> dict[str, Any]: + """Accept one legacy result object or one unambiguous result event.""" + + message = f"{display_name} {operation} returned an unexpected response." + if isinstance(payload, dict): + if "type" in payload and payload.get("type") != "result": + raise AdvisorError(message) + if "subtype" in payload and payload.get("subtype") != "success": + raise AdvisorError(message) + return payload + if not isinstance(payload, list) or not payload: + raise AdvisorError(message) + if not all( + isinstance(event, dict) + and isinstance(event.get("type"), str) + and bool(event["type"]) + for event in payload + ): + raise AdvisorError(message) + result_events = [event for event in payload if event.get("type") == "result"] + if len(result_events) != 1: + raise AdvisorError(message) + selected = result_events[0] + if selected.get("subtype") not in (None, "success"): + raise AdvisorError(message) + content_fields = {"result", "modelUsage", "structured_output"} + if any( + event is not selected and content_fields.intersection(event) + for event in payload + ): + raise AdvisorError(message) + return selected + + +def _validate_review_output(value: Any, *, display_name: str) -> dict[str, str]: + if not isinstance(value, dict) or set(value) != set(PLAN_REVIEW_SCHEMA["required"]): + raise AdvisorError( + f"{display_name} plan review returned invalid structured output." + ) + signal = value.get("signal") + body = value.get("body") + if ( + not isinstance(signal, str) + or signal not in {"PLAN_APPROVED", "PLAN_REVISE"} + or not isinstance(body, str) + or not body.strip() + ): + raise AdvisorError( + f"{display_name} plan review returned invalid structured output." + ) + return {"signal": signal, "body": body.strip()} + + +def _review_response(payload: dict[str, Any], *, display_name: str) -> tuple[str, str]: + structured_present = "structured_output" in payload + result_present = "result" in payload + structured = ( + _validate_review_output( + payload.get("structured_output"), + display_name=display_name, + ) + if structured_present + else None + ) + legacy: dict[str, str] | None = None + if result_present: + raw_result = payload.get("result") + if not isinstance(raw_result, str): + raise AdvisorError( + f"{display_name} plan review returned invalid structured output." + ) + try: + decoded_result = json.loads(raw_result) + except json.JSONDecodeError: + if structured is None: + raise AdvisorError( + f"{display_name} plan review returned invalid structured output." + ) + else: + legacy = _validate_review_output( + decoded_result, + display_name=display_name, + ) + if structured is None and legacy is None: + raise AdvisorError( + f"{display_name} plan review returned invalid structured output." + ) + if structured is not None and legacy is not None and structured != legacy: + raise AdvisorError( + f"{display_name} plan review returned conflicting structured output." + ) + selected = structured or legacy + assert selected is not None + return selected["signal"], f"{selected['signal']}\n{selected['body']}" + + def _invoke_fable( *, operation: str, @@ -397,6 +540,17 @@ def _invoke_fable( "--system-prompt", system_prompt, ] + if operation == "plan review": + command.extend( + ( + "--json-schema", + json.dumps( + PLAN_REVIEW_SCHEMA, + separators=(",", ":"), + sort_keys=True, + ), + ) + ) try: result = subprocess.run( command, @@ -417,19 +571,33 @@ def _invoke_fable( f"{display_name} {operation} exited with {result.returncode}; output withheld." ) try: - payload = json.loads(result.stdout) + decoded = json.loads(result.stdout) except json.JSONDecodeError as exc: raise AdvisorError(f"{display_name} {operation} returned malformed JSON.") from exc - if not isinstance(payload, dict) or not isinstance(payload.get("result"), str): - raise AdvisorError(f"{display_name} {operation} returned an unexpected response.") + payload = _normalize_model_payload( + decoded, + display_name=display_name, + operation=operation, + ) # Authorize the complete runtime identity set before interpreting or # returning any model-authored plan/review content. used_models = _validate_runtime_models(payload.get("modelUsage"), route["model"]) - response = payload["result"].strip() - signal = _first_non_empty_line(response) + if operation == "plan review": + signal, response = _review_response(payload, display_name=display_name) + else: + if "structured_output" in payload or not isinstance( + payload.get("result"), str + ): + raise AdvisorError( + f"{display_name} {operation} returned an unexpected response." + ) + response = payload["result"].strip() + signal = _first_non_empty_line(response) if signal not in allowed_signals: if operation == "plan review": - raise AdvisorError(f"{display_name} omitted the required plan decision.") + raise AdvisorError( + f"{display_name} returned an invalid structured plan decision." + ) expected = " or ".join(sorted(allowed_signals)) raise AdvisorError( f"{display_name} {operation} omitted the required {expected} signal." diff --git a/tests/plugin_lifecycle_smoke.py b/tests/plugin_lifecycle_smoke.py index 7bd20b0..bb6d15f 100755 --- a/tests/plugin_lifecycle_smoke.py +++ b/tests/plugin_lifecycle_smoke.py @@ -32,7 +32,7 @@ MARKETPLACE_NAME = "codex-orchestration" OLD_RELEASE = "a1d9c546665c3253cdcaa8fe5c0c060199a6126c" OLD_VERSION = "0.5.0" -NEW_VERSION = "0.9.0" +NEW_VERSION = "0.9.1" COMMAND_TIMEOUT_SECONDS = 60 diff --git a/tests/test_external_subscription.py b/tests/test_external_subscription.py index dde52a3..d83b214 100644 --- a/tests/test_external_subscription.py +++ b/tests/test_external_subscription.py @@ -85,6 +85,35 @@ def test_invoke_preserves_existing_no_tools_bridge_and_runtime_identity(self) -> create.assert_called_once_with(packet="bounded planning packet") self.assertIs(result, expected) + def test_fable_invoke_accepts_either_reviewed_primary_runtime_identity( + self, + ) -> None: + for primary in ("claude-fable-5", "claude-opus-4-8"): + expected = { + "model": "claude-fable-5", + "effort": "high", + "used_models": [ + primary, + SUBSCRIPTION.fable_advisor_mcp.FABLE_HELPER_MODEL, + ], + "signal": "PLAN_DRAFT", + } + with self.subTest(primary=primary), mock.patch.object( + SUBSCRIPTION.fable_advisor_mcp, + "load_fable_route", + return_value={"model": "claude-fable-5", "effort": "high"}, + ), mock.patch.object( + SUBSCRIPTION.fable_advisor_mcp, + "create_plan", + return_value=expected, + ): + self.assertIs( + SUBSCRIPTION.invoke( + "create_plan", {"packet": "bounded planning packet"} + ), + expected, + ) + def test_argument_shape_and_runtime_metadata_fail_closed(self) -> None: with self.assertRaisesRegex( SUBSCRIPTION.SubscriptionAdapterError, "arguments" diff --git a/tests/test_fable_advisor_mcp.py b/tests/test_fable_advisor_mcp.py index ba5acf5..214b41c 100644 --- a/tests/test_fable_advisor_mcp.py +++ b/tests/test_fable_advisor_mcp.py @@ -27,6 +27,8 @@ FABLE = importlib.util.module_from_spec(SPEC) SPEC.loader.exec_module(FABLE) DEFAULT_MODEL_USAGE = object() +DEFAULT_STRUCTURED_OUTPUT = object() +AUTO_STRUCTURED_OUTPUT = object() class FableAdvisorMcpTests(unittest.TestCase): @@ -115,7 +117,9 @@ def completed( ) -> subprocess.CompletedProcess[str]: return subprocess.CompletedProcess(command, returncode, stdout, stderr) - def auth_result(self) -> subprocess.CompletedProcess[str]: + def auth_result( + self, subscription_type: object = "max" + ) -> subprocess.CompletedProcess[str]: return self.completed( ["claude", "auth", "status"], json.dumps( @@ -123,24 +127,38 @@ def auth_result(self) -> subprocess.CompletedProcess[str]: "loggedIn": True, "authMethod": "claude.ai", "apiProvider": "firstParty", - "subscriptionType": "max", + "subscriptionType": subscription_type, } ), ) def model_result( - self, response: str, *, model_usage: object = DEFAULT_MODEL_USAGE + self, + response: str, + *, + model_usage: object = DEFAULT_MODEL_USAGE, + structured_output: object = DEFAULT_STRUCTURED_OUTPUT, + as_events: bool = False, ) -> subprocess.CompletedProcess[str]: + payload: dict[str, object] = { + "result": response, + "modelUsage": model_usage + if model_usage is not DEFAULT_MODEL_USAGE + else {"claude-fable-5": {"outputTokens": 12}}, + } + if structured_output is not DEFAULT_STRUCTURED_OUTPUT: + payload["structured_output"] = structured_output + outer: object = ( + [ + {"type": "system", "subtype": "init"}, + {"type": "result", "subtype": "success", **payload}, + ] + if as_events + else payload + ) return self.completed( ["claude"], - json.dumps( - { - "result": response, - "modelUsage": model_usage - if model_usage is not DEFAULT_MODEL_USAGE - else {"claude-fable-5": {"outputTokens": 12}}, - } - ), + json.dumps(outer), ) def invoke_with_results( @@ -149,6 +167,8 @@ def invoke_with_results( *args: str, model_response: str, model_usage: object = DEFAULT_MODEL_USAGE, + structured_output: object = AUTO_STRUCTURED_OUTPUT, + as_events: bool = False, ) -> tuple[dict[str, object], list[tuple[list[str], dict[str, object]]]]: calls: list[tuple[list[str], dict[str, object]]] = [] @@ -156,9 +176,37 @@ def fake_run( command: list[str], **kwargs: object ) -> subprocess.CompletedProcess[str]: calls.append((command, kwargs)) - if command[-2:] == ["auth", "status"]: + if command[-2:] == ["auth", "status"] or command[-3:] == [ + "auth", + "status", + "--json", + ]: return self.auth_result() - return self.model_result(model_response, model_usage=model_usage) + selected_structured_output = structured_output + if ( + selected_structured_output is AUTO_STRUCTURED_OUTPUT + and function is FABLE.review_plan + ): + lines = model_response.strip().splitlines() + if ( + lines + and lines[0] in {"PLAN_APPROVED", "PLAN_REVISE"} + and "\n".join(lines[1:]).strip() + ): + selected_structured_output = { + "signal": lines[0], + "body": "\n".join(lines[1:]).strip(), + } + else: + selected_structured_output = DEFAULT_STRUCTURED_OUTPUT + elif selected_structured_output is AUTO_STRUCTURED_OUTPUT: + selected_structured_output = DEFAULT_STRUCTURED_OUTPUT + return self.model_result( + model_response, + model_usage=model_usage, + structured_output=selected_structured_output, + as_events=as_events, + ) with ( mock.patch.dict(os.environ, {"CODEX_HOME": str(self.home)}), @@ -170,6 +218,25 @@ def fake_run( result = function(*args) return result, calls + def invoke_with_stdout( + self, function: object, *args: str, stdout: str + ) -> dict[str, object]: + with ( + mock.patch.dict(os.environ, {"CODEX_HOME": str(self.home)}), + mock.patch.object( + FABLE, "resolve_claude", return_value=Path("/fake/claude") + ), + mock.patch.object( + FABLE.subprocess, + "run", + side_effect=[ + self.auth_result(), + self.completed(["claude"], stdout), + ], + ), + ): + return function(*args) + def test_review_is_pinned_sanitized_read_only_and_runtime_confirmed(self) -> None: env = { "CODEX_HOME": str(self.home), @@ -181,9 +248,24 @@ def fake_run( command: list[str], **kwargs: object ) -> subprocess.CompletedProcess[str]: calls.append((command, kwargs)) - if command[-2:] == ["auth", "status"]: + if command[-2:] == ["auth", "status"] or command[-3:] == [ + "auth", + "status", + "--json", + ]: return self.auth_result() - return self.model_result("PLAN_APPROVED\nNo material gap found.") + return self.model_result( + json.dumps( + { + "signal": "PLAN_APPROVED", + "body": "No material gap found.", + } + ), + structured_output={ + "signal": "PLAN_APPROVED", + "body": "No material gap found.", + }, + ) with ( mock.patch.dict(os.environ, env, clear=False), @@ -199,7 +281,7 @@ def fake_run( self.assertEqual(result["used_models"], ["claude-fable-5"]) self.assertNotIn("subscription_type", result) auth_command, auth_kwargs = calls[0] - self.assertEqual(auth_command[-2:], ["auth", "status"]) + self.assertEqual(auth_command[-3:], ["auth", "status", "--json"]) review_command, review_kwargs = calls[1] for flag in ( "--print", @@ -209,6 +291,7 @@ def fake_run( "--no-session-persistence", "--prompt-suggestions", "--output-format", + "--json-schema", "--system-prompt", ): self.assertIn(flag, review_command) @@ -228,6 +311,10 @@ def fake_run( self.assertEqual( review_command[review_command.index("--output-format") + 1], "json" ) + self.assertEqual( + json.loads(review_command[review_command.index("--json-schema") + 1]), + FABLE.PLAN_REVIEW_SCHEMA, + ) self.assertEqual(review_kwargs["input"], "Review this complete plan.") for kwargs in (auth_kwargs, review_kwargs): sanitized = kwargs["env"] @@ -265,6 +352,8 @@ def test_auth_and_model_subprocesses_receive_only_platform_runtime_environment( "LC_CTYPE": "UTF-8", "HOME": "/trusted/home", "TMPDIR": "/trusted/tmp", + "USER": "hostile-user", + "LOGNAME": "hostile-logname", "SystemRoot": r"C:\should-not-pass", **hostile, }, @@ -275,6 +364,8 @@ def test_auth_and_model_subprocesses_receive_only_platform_runtime_environment( "LC_CTYPE": "UTF-8", "HOME": "/trusted/home", "TMPDIR": "/trusted/tmp", + "USER": "trusted-user", + "LOGNAME": "trusted-user", "CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC": "1", }, ), @@ -318,13 +409,33 @@ def fake_run( command: list[str], **kwargs: object ) -> subprocess.CompletedProcess[str]: calls.append((command, kwargs)) - if command[-2:] == ["auth", "status"]: + if command[-2:] == ["auth", "status"] or command[-3:] == [ + "auth", + "status", + "--json", + ]: return self.auth_result() - return self.model_result("PLAN_APPROVED\nNo material gap found.") + return self.model_result( + json.dumps( + { + "signal": "PLAN_APPROVED", + "body": "No material gap found.", + } + ), + structured_output={ + "signal": "PLAN_APPROVED", + "body": "No material gap found.", + }, + ) with ( mock.patch.dict(os.environ, inherited, clear=True), mock.patch.object(FABLE.os, "name", platform), + mock.patch.object( + FABLE, + "_canonical_posix_identity", + return_value="trusted-user", + ), mock.patch.object( FABLE, "load_fable_route", @@ -340,17 +451,135 @@ def fake_run( for _, kwargs in calls: self.assertEqual(kwargs["env"], expected) + def test_auth_accepts_only_exact_first_party_pro_max_or_team_tuples(self) -> None: + valid_subscriptions = ("pro", "max", "team") + executable = Path("/fake/claude") + for subscription in valid_subscriptions: + with self.subTest(subscription=subscription), mock.patch.object( + FABLE, + "_run_json", + return_value={ + "loggedIn": True, + "authMethod": "claude.ai", + "apiProvider": "firstParty", + "subscriptionType": subscription, + }, + ) as run: + self.assertEqual( + FABLE.check_claude_auth(executable), + { + "auth_method": "claude.ai", + "api_provider": "firstParty", + }, + ) + self.assertEqual( + run.call_args.args[0], + [str(executable), "auth", "status", "--json"], + ) + + invalid_payloads = ( + { + "loggedIn": False, + "authMethod": "claude.ai", + "apiProvider": "firstParty", + "subscriptionType": "team", + }, + { + "loggedIn": 1, + "authMethod": "claude.ai", + "apiProvider": "firstParty", + "subscriptionType": "team", + }, + { + "loggedIn": True, + "authMethod": "console", + "apiProvider": "firstParty", + "subscriptionType": "team", + }, + { + "loggedIn": True, + "authMethod": "claude.ai", + "apiProvider": "bedrock", + "subscriptionType": "team", + }, + { + "loggedIn": True, + "authMethod": "claude.ai", + "apiProvider": "firstParty", + "subscriptionType": "Team", + }, + { + "loggedIn": True, + "authMethod": "claude.ai", + "apiProvider": "firstParty", + "subscriptionType": "enterprise", + }, + { + "loggedIn": True, + "authMethod": "claude.ai", + "apiProvider": "firstParty", + "subscriptionType": [], + }, + { + "loggedIn": True, + "authMethod": "claude.ai", + "apiProvider": "firstParty", + }, + ) + secret = "TOP-SECRET-AUTH-METADATA" + for payload in invalid_payloads: + with self.subTest(payload=payload), mock.patch.object( + FABLE, "_run_json", return_value={**payload, "account": secret} + ): + with self.assertRaises(FABLE.AdvisorError) as failure: + FABLE.check_claude_auth(executable) + self.assertIn("Pro, Max, or Team", str(failure.exception)) + self.assertNotIn(secret, str(failure.exception)) + + def test_posix_identity_lookup_failure_stops_before_any_subprocess(self) -> None: + executable = Path("/fake/claude") + with ( + mock.patch.object(FABLE.os, "name", "posix"), + mock.patch.object( + FABLE, + "_canonical_posix_identity", + side_effect=FABLE.AdvisorError("canonical identity unavailable"), + ), + mock.patch.object(FABLE.subprocess, "run") as run, + ): + with self.assertRaisesRegex( + FABLE.AdvisorError, "canonical identity unavailable" + ): + FABLE.check_claude_auth(executable) + run.assert_not_called() + def test_runtime_model_policy_accepts_only_fable_and_exact_allowed_helper( self, ) -> None: + resolved_primary = "claude-opus-4-8" allowed_scenarios = ( ({FABLE.FABLE_MODEL: {"outputTokens": 12}}, [FABLE.FABLE_MODEL]), + ({resolved_primary: {"outputTokens": 12}}, [resolved_primary]), + ( + { + resolved_primary: {"outputTokens": 12}, + FABLE.FABLE_HELPER_MODEL: {"outputTokens": 1}, + }, + sorted((resolved_primary, FABLE.FABLE_HELPER_MODEL)), + ), ( { FABLE.FABLE_MODEL: {"outputTokens": 12}, + resolved_primary: {"outputTokens": 12}, FABLE.FABLE_HELPER_MODEL: {"outputTokens": 1}, }, - sorted((FABLE.FABLE_MODEL, FABLE.FABLE_HELPER_MODEL)), + sorted( + ( + FABLE.FABLE_MODEL, + resolved_primary, + FABLE.FABLE_HELPER_MODEL, + ) + ), ), ) for model_usage, expected_models in allowed_scenarios: @@ -759,6 +988,164 @@ def test_repeated_revisions_are_fresh_and_never_use_sessions(self) -> None: self.assertNotIn("--resume", command) self.assertNotIn("--session-id", command) + def test_review_uses_and_locally_enforces_the_exact_structured_schema(self) -> None: + structured = { + "signal": "PLAN_APPROVED", + "body": "No material gap found.", + } + result, calls = self.invoke_with_results( + FABLE.review_plan, + "packet", + model_response="This prose is not the decision contract.", + structured_output=structured, + ) + self.assertEqual(result["decision"], "PLAN_APPROVED") + self.assertEqual(result["review"], "PLAN_APPROVED\nNo material gap found.") + command = calls[1][0] + self.assertEqual(command.count("--json-schema"), 1) + self.assertEqual( + json.loads(command[command.index("--json-schema") + 1]), + { + "type": "object", + "properties": { + "signal": { + "type": "string", + "enum": ["PLAN_APPROVED", "PLAN_REVISE"], + }, + "body": {"type": "string", "minLength": 1}, + }, + "required": ["signal", "body"], + "additionalProperties": False, + }, + ) + + legacy, _ = self.invoke_with_results( + FABLE.review_plan, + "packet", + model_response=json.dumps( + { + "signal": "PLAN_REVISE", + "body": "F-1: add the missing negative regression.", + } + ), + ) + self.assertEqual(legacy["decision"], "PLAN_REVISE") + self.assertEqual( + legacy["review"], + "PLAN_REVISE\nF-1: add the missing negative regression.", + ) + + malformed = ( + ("PLAN_APPROVED\nraw prose is not structured", DEFAULT_STRUCTURED_OUTPUT), + (json.dumps({"signal": "PLAN_APPROVED"}), DEFAULT_STRUCTURED_OUTPUT), + ( + json.dumps({"signal": "PLAN_APPROVED", "body": "ok", "extra": 1}), + DEFAULT_STRUCTURED_OUTPUT, + ), + ( + json.dumps({"signal": "PLAN_DRAFT", "body": "wrong signal"}), + DEFAULT_STRUCTURED_OUTPUT, + ), + ( + json.dumps({"signal": "PLAN_APPROVED", "body": " "}), + DEFAULT_STRUCTURED_OUTPUT, + ), + ( + json.dumps({"signal": "PLAN_APPROVED", "body": 7}), + DEFAULT_STRUCTURED_OUTPUT, + ), + ( + json.dumps({"signal": "PLAN_APPROVED", "body": "one"}), + {"signal": "PLAN_REVISE", "body": "two"}, + ), + ) + secret = "TOP-SECRET-STRUCTURED-OUTPUT" + for response, structured_output in malformed: + with self.subTest(response=response, structured=structured_output): + with self.assertRaises(FABLE.AdvisorError) as failure: + self.invoke_with_results( + FABLE.review_plan, + "packet", + model_response=response.replace("raw prose", secret), + structured_output=structured_output, + ) + self.assertNotIn(secret, str(failure.exception)) + + def test_cli_output_container_accepts_one_result_and_rejects_ambiguity( + self, + ) -> None: + self.write_state(planner=self.route()) + created, _ = self.invoke_with_results( + FABLE.create_plan, + "packet", + model_response="PLAN_DRAFT\nDraft", + as_events=True, + ) + self.assertEqual(created["signal"], "PLAN_DRAFT") + + revision = ( + "PLAN_REVISION\n## FINDINGS_LEDGER\n" + "F-1 INCORPORATED: fixed.\n## REVISED_PLAN\nv2" + ) + revised, _ = self.invoke_with_results( + FABLE.revise_plan, + "task", + "v1", + "F-1", + "history", + model_response=revision, + as_events=True, + ) + self.assertEqual(revised["signal"], "PLAN_REVISION") + + self.write_state(advisor=self.route()) + reviewed, _ = self.invoke_with_results( + FABLE.review_plan, + "packet", + model_response="ignored prose", + structured_output={ + "signal": "PLAN_APPROVED", + "body": "No material gap.", + }, + as_events=True, + ) + self.assertEqual(reviewed["decision"], "PLAN_APPROVED") + + result_event = { + "type": "result", + "subtype": "success", + "result": json.dumps( + {"signal": "PLAN_APPROVED", "body": "No material gap."} + ), + "modelUsage": {FABLE.FABLE_MODEL: {"outputTokens": 12}}, + } + secret = "TOP-SECRET-AMBIGUOUS-EVENT" + malformed_outers: tuple[object, ...] = ( + [], + [{"type": "system", "subtype": "init"}], + [result_event, result_event], + [{"type": "system"}, secret, result_event], + {**result_event, "type": "assistant"}, + {**result_event, "type": None}, + { + "subtype": "error", + "result": json.dumps( + {"signal": "PLAN_APPROVED", "body": secret} + ), + "modelUsage": {FABLE.FABLE_MODEL: {"outputTokens": 12}}, + }, + secret, + ) + for outer in malformed_outers: + with self.subTest(outer=outer): + with self.assertRaises(FABLE.AdvisorError) as failure: + self.invoke_with_stdout( + FABLE.review_plan, + "packet", + stdout=json.dumps(outer), + ) + self.assertNotIn(secret, str(failure.exception)) + def test_malformed_json_unconfirmed_model_and_bad_review_fail_closed(self) -> None: bad_outputs = ( ("not json", "malformed JSON"), @@ -788,7 +1175,7 @@ def test_malformed_json_unconfirmed_model_and_bad_review_fail_closed(self) -> No FABLE.create_plan("packet") self.write_state(advisor=self.route()) - with self.assertRaisesRegex(FABLE.AdvisorError, "required plan decision"): + with self.assertRaisesRegex(FABLE.AdvisorError, "structured output"): self.invoke_with_results( FABLE.review_plan, "packet", model_response="Looks good." ) diff --git a/tests/test_native_routing.py b/tests/test_native_routing.py index fa009ff..04f98b0 100644 --- a/tests/test_native_routing.py +++ b/tests/test_native_routing.py @@ -298,7 +298,7 @@ def setUp(self) -> None: #!/usr/bin/env python3 import json import sys - if sys.argv[1:] == ["auth", "status"]: + if sys.argv[1:] in (["auth", "status"], ["auth", "status", "--json"]): print(json.dumps({ "loggedIn": True, "authMethod": "claude.ai", @@ -312,7 +312,7 @@ def setUp(self) -> None: "(low, medium, high, xhigh, max) " "--safe-mode --tools --permission-mode " "--no-session-persistence --prompt-suggestions " - "--output-format --system-prompt" + "--output-format --json-schema --system-prompt" ) raise SystemExit(0) if sys.argv[1:] == ["--version"]: @@ -2432,6 +2432,7 @@ def test_bundled_claude_prerequisite_checks_every_runtime_control(self) -> None: "--no-session-persistence", "--prompt-suggestions", "--output-format", + "--json-schema", "--system-prompt", ) for flag in required: @@ -2540,6 +2541,7 @@ def test_bundled_claude_requires_exact_long_option_tokens(self) -> None: "--no-session-persistence", "--prompt-suggestions", "--output-format", + "--json-schema", "--system-prompt", ) for flag in required: @@ -2669,7 +2671,7 @@ def test_require_effective_rejects_unavailable_fable_auth(self) -> None: self.assertEqual(status.returncode, 1) self.assertIn( - "must be logged in through a first-party Pro or Max account", + "must be logged in through a first-party Pro, Max, or Team account", status.stdout, ) diff --git a/tests/test_packaging.py b/tests/test_packaging.py index 9e4831b..92f3518 100644 --- a/tests/test_packaging.py +++ b/tests/test_packaging.py @@ -255,7 +255,7 @@ def test_plugin_marketplace_and_skill_names_are_aligned(self) -> None: self.assertEqual(manifest["name"], "codex-orchestration") self.assertEqual(manifest["skills"], "./skills/") - self.assertEqual(manifest["version"], "0.9.0") + self.assertEqual(manifest["version"], "0.9.1") self.assertEqual(manifest["mcpServers"], "./.mcp.json") self.assertRegex( manifest["version"], @@ -278,7 +278,7 @@ def test_native_and_custom_configurators_are_packaged(self) -> None: self.assertFalse((SKILL_ROOT / "scripts" / "update_plugin.py").exists()) self.assertIn("config/batchWrite", native.read_text(encoding="utf-8")) self.assertIn('"--repair"', native.read_text(encoding="utf-8")) - self.assertIn('"version": "0.9.0"', native.read_text(encoding="utf-8")) + self.assertIn('"version": "0.9.1"', native.read_text(encoding="utf-8")) self.assertIn("validate_routing_state", routing_state.read_text(encoding="utf-8")) self.assertIn("Standalone custom agent", custom.read_text(encoding="utf-8")) @@ -389,7 +389,7 @@ def test_ci_runs_dual_version_plugin_lifecycle(self) -> None: self.assertIn("@openai/codex@0.144.1", workflow) smoke_text = smoke.read_text(encoding="utf-8") self.assertIn('OLD_VERSION = "0.5.0"', smoke_text) - self.assertIn('NEW_VERSION = "0.9.0"', smoke_text) + self.assertIn('NEW_VERSION = "0.9.1"', smoke_text) self.assertIn("old Advisor-only cache unexpectedly supports Planner", smoke_text) self.assertIn("Upgraded installed skill is missing Planner contract", smoke_text) self.assertIn("reused the Advisor-only 0.5.0 cache directory", smoke_text) diff --git a/tests/test_release_check.py b/tests/test_release_check.py index 0867a73..b7ee97d 100644 --- a/tests/test_release_check.py +++ b/tests/test_release_check.py @@ -69,7 +69,7 @@ def commit(self, message: str) -> str: class ReleaseCheckTests(unittest.TestCase): def test_checkout_release_metadata_is_consistent(self) -> None: - self.assertEqual(RELEASE.run_check(REPO_ROOT, require_tag=False), "0.9.0") + self.assertEqual(RELEASE.run_check(REPO_ROOT, require_tag=False), "0.9.1") def test_unreleased_checkout_is_not_tag_ready(self) -> None: with self.assertRaisesRegex(RELEASE.ReleaseCheckError, "not tagged"): diff --git a/tests/test_skill_contract.py b/tests/test_skill_contract.py index 0e0e2e1..9ad0831 100644 --- a/tests/test_skill_contract.py +++ b/tests/test_skill_contract.py @@ -301,9 +301,13 @@ def test_fable_is_a_bundled_root_only_planner_or_advisor(self) -> None: self.assertIn("--planner-fable --planner-effort ", SKILL) self.assertIn("built-in cross-provider Planner or Advisor exception", SKILL) self.assertIn("All bundled variants are disabled by default", SKILL) - self.assertIn("first-party Pro or Max account", SKILL) + self.assertIn("first-party Pro, Max, or Team account", SKILL) self.assertIn("never extracts a token", SKILL) - self.assertIn("runtime `modelUsage` to contain the pinned `claude-fable-5`", SKILL) + self.assertIn( + "reviewed Fable primary identity (`claude-fable-5` or " + "`claude-opus-4-8`)", + SKILL, + ) self.assertIn("explicit exact helper allowlist", SKILL) self.assertIn("unknown additional or missing primary model", SKILL) self.assertIn("`create_plan`", SKILL)