diff --git a/.deps.lock.json b/.deps.lock.json index bb2c9c36..1c9f6a3d 100644 --- a/.deps.lock.json +++ b/.deps.lock.json @@ -38,6 +38,7 @@ "dependencies": [ "cryptography>=50.0.0,<51.0.0", "logion-client", + "logion-runner", "logion-skillmap", "pydantic>=2.7,<3.0.0", "pyyaml>=6.0,<7.0", diff --git a/.github/workflows/pr-safety.yml b/.github/workflows/pr-safety.yml index 691c523e..bb8808d1 100644 --- a/.github/workflows/pr-safety.yml +++ b/.github/workflows/pr-safety.yml @@ -274,18 +274,8 @@ jobs: - name: Set up Rust toolchain uses: dtolnay/rust-toolchain@stable - - name: Install sf - run: > - cargo install --git https://github.com/nicolasmelo1/software-factory - --rev b06be44f6c982dac58b898778d7dba224d9ed7b1 --locked - - # See the comment on `factory-check` in the Makefile for why the - # flag is required rather than a convenience. - - name: Prove the factory rules still fire - run: sf verify --allow-commands - - - name: Software-factory rules - run: sf check --allow-commands + - name: Prove and check the pinned factory rules + run: make factory-check - name: Installer security guardrails run: make check-installer-security diff --git a/.secrets.baseline b/.secrets.baseline index 63a144f1..e7887234 100644 --- a/.secrets.baseline +++ b/.secrets.baseline @@ -127,15 +127,6 @@ } ], "results": { - "Makefile": [ - { - "type": "Secret Keyword", - "filename": "Makefile", - "hashed_secret": "e441659cd1b30148c99024a81228d16b63f772af", - "is_verified": false, - "line_number": 51 - } - ], "artifacts/phase-gates/phase-15.11.json": [ { "type": "Hex High Entropy String", @@ -200,14 +191,14 @@ { "type": "Hex High Entropy String", "filename": "artifacts/phase-gates/phase-15.14.1.json", - "hashed_secret": "6572bfe78cf0936d82bea1e92cb038976838fceb", + "hashed_secret": "32b12c3bcdec465b01ec886bcceb7b373ddc94be", "is_verified": false, "line_number": 10 }, { "type": "Hex High Entropy String", "filename": "artifacts/phase-gates/phase-15.14.1.json", - "hashed_secret": "03603e0736d88c4dceb18e34dd7afc67ec62ba48", + "hashed_secret": "edb9ca5a7f371f55f2494930a4f2e9195f83a383", "is_verified": false, "line_number": 14 } @@ -223,14 +214,37 @@ { "type": "Hex High Entropy String", "filename": "artifacts/phase-gates/phase-15.15.json", - "hashed_secret": "7c107851c0bcd0a5565a8f0f9710f5588913c6d4", + "hashed_secret": "3a7ff8f9d83b3efdf386af73d796a9bdeb749d4d", "is_verified": false, "line_number": 10 }, { "type": "Hex High Entropy String", "filename": "artifacts/phase-gates/phase-15.15.json", - "hashed_secret": "46e16693d39796f2f1b4a5b0df96ca695e41c14e", + "hashed_secret": "eb117f4eb95750299f1a4a09c3aeffb20794ab87", + "is_verified": false, + "line_number": 14 + } + ], + "artifacts/phase-gates/phase-16.1.json": [ + { + "type": "Hex High Entropy String", + "filename": "artifacts/phase-gates/phase-16.1.json", + "hashed_secret": "718fa2c8291bcc2f2bdd7a48e786b1edd73aedb4", + "is_verified": false, + "line_number": 9 + }, + { + "type": "Hex High Entropy String", + "filename": "artifacts/phase-gates/phase-16.1.json", + "hashed_secret": "e091c0c4b2dc131d7a22e48b08c8572051712333", + "is_verified": false, + "line_number": 10 + }, + { + "type": "Hex High Entropy String", + "filename": "artifacts/phase-gates/phase-16.1.json", + "hashed_secret": "eb8f5d6f907a50e99e1e4d80449940bfc206c7d6", "is_verified": false, "line_number": 14 } @@ -248,7 +262,7 @@ { "type": "Hex High Entropy String", "filename": "artifacts/phase-gates/reports/phase-15.14.1-local-multi-agent-node.json", - "hashed_secret": "b63228c6fd53e2e2b37df3fff8a8566ec48b65df", + "hashed_secret": "6351ff613175a96f4fd4d60e6bb6a46886c7006f", "is_verified": false, "line_number": 35 }, @@ -262,7 +276,7 @@ { "type": "Hex High Entropy String", "filename": "artifacts/phase-gates/reports/phase-15.14.1-local-multi-agent-node.json", - "hashed_secret": "06312def5a7290275fd3b11d78c56c21d35e55af", + "hashed_secret": "3b6da7224759d10ff2dabc040bd3d8c594f97598", "is_verified": false, "line_number": 42 }, @@ -274,6 +288,57 @@ "line_number": 43 } ], + "artifacts/phase-gates/reports/phase-15.15-isolated-runner-node.json": [ + { + "type": "Hex High Entropy String", + "filename": "artifacts/phase-gates/reports/phase-15.15-isolated-runner-node.json", + "hashed_secret": "c690878c7e96a6f55528e285135450e8de29ad1f", + "is_verified": false, + "line_number": 66 + }, + { + "type": "Hex High Entropy String", + "filename": "artifacts/phase-gates/reports/phase-15.15-isolated-runner-node.json", + "hashed_secret": "ed85ced3fd5a22f15d26f723bc88e1b0df16336d", + "is_verified": false, + "line_number": 101 + }, + { + "type": "Hex High Entropy String", + "filename": "artifacts/phase-gates/reports/phase-15.15-isolated-runner-node.json", + "hashed_secret": "d95155ec5c8e812290e4089c895fd0ae31911ae7", + "is_verified": false, + "line_number": 124 + }, + { + "type": "Hex High Entropy String", + "filename": "artifacts/phase-gates/reports/phase-15.15-isolated-runner-node.json", + "hashed_secret": "cc028acd985cad9c0babbe8ed0a5dccdf2ab87d0", + "is_verified": false, + "line_number": 155 + }, + { + "type": "Hex High Entropy String", + "filename": "artifacts/phase-gates/reports/phase-15.15-isolated-runner-node.json", + "hashed_secret": "6d9e7bf1a2e277c438a0927614b69b75fdc1639b", + "is_verified": false, + "line_number": 167 + }, + { + "type": "Hex High Entropy String", + "filename": "artifacts/phase-gates/reports/phase-15.15-isolated-runner-node.json", + "hashed_secret": "e7a8c6f2bccf79d68414a1f88b4c9fca0e92bb7e", + "is_verified": false, + "line_number": 177 + }, + { + "type": "Secret Keyword", + "filename": "artifacts/phase-gates/reports/phase-15.15-isolated-runner-node.json", + "hashed_secret": "5f5f8758f5f22d523e531f58123b6db9161683a4", + "is_verified": false, + "line_number": 262 + } + ], "artifacts/phase-gates/reports/phase-15.15-isolated-runner.json": [ { "type": "Hex High Entropy String", @@ -318,6 +383,50 @@ "line_number": 236 } ], + "artifacts/phase-gates/reports/phase-16.1-eval-contract-reference-runner.json": [ + { + "type": "Hex High Entropy String", + "filename": "artifacts/phase-gates/reports/phase-16.1-eval-contract-reference-runner.json", + "hashed_secret": "a963a06acd4e98b545d8624ce4623c16ff9b898c", + "is_verified": false, + "line_number": 59 + }, + { + "type": "Hex High Entropy String", + "filename": "artifacts/phase-gates/reports/phase-16.1-eval-contract-reference-runner.json", + "hashed_secret": "d4b46cd6d3b81fe6d83d7feee377e0e283a35bbb", + "is_verified": false, + "line_number": 95 + }, + { + "type": "Hex High Entropy String", + "filename": "artifacts/phase-gates/reports/phase-16.1-eval-contract-reference-runner.json", + "hashed_secret": "ccf3851c628aa4bd4c8e59aaa827cdfb5f90b893", + "is_verified": false, + "line_number": 96 + }, + { + "type": "Hex High Entropy String", + "filename": "artifacts/phase-gates/reports/phase-16.1-eval-contract-reference-runner.json", + "hashed_secret": "908486ea6aec643f5f529814f033c817534c3a20", + "is_verified": false, + "line_number": 116 + }, + { + "type": "Hex High Entropy String", + "filename": "artifacts/phase-gates/reports/phase-16.1-eval-contract-reference-runner.json", + "hashed_secret": "2d64c03330871bbb503077102297c243a8eb5492", + "is_verified": false, + "line_number": 130 + }, + { + "type": "Hex High Entropy String", + "filename": "artifacts/phase-gates/reports/phase-16.1-eval-contract-reference-runner.json", + "hashed_secret": "6a442408973c7d61e8771e10bf2d780aa05a9a4e", + "is_verified": false, + "line_number": 143 + } + ], "artifacts/phase-gates/reports/remote-private-mcp-feedback.json": [ { "type": "Hex High Entropy String", @@ -582,7 +691,7 @@ { "type": "Hex High Entropy String", "filename": "packages/eval-contract/tests/fixtures/golden_contract.json", - "hashed_secret": "8e313bdea1a5847f37b08c2ca4dd3648537405cc", + "hashed_secret": "908486ea6aec643f5f529814f033c817534c3a20", "is_verified": false, "line_number": 12 } @@ -591,7 +700,7 @@ { "type": "Hex High Entropy String", "filename": "packages/eval-contract/tests/fixtures/golden_contract.yaml", - "hashed_secret": "8e313bdea1a5847f37b08c2ca4dd3648537405cc", + "hashed_secret": "908486ea6aec643f5f529814f033c817534c3a20", "is_verified": false, "line_number": 12 } @@ -830,7 +939,16 @@ "is_verified": false, "line_number": 75 } + ], + "scripts/sf.py": [ + { + "type": "Hex High Entropy String", + "filename": "scripts/sf.py", + "hashed_secret": "205a5a191149dca0f898bb7c98838e6e19f67f70", + "is_verified": false, + "line_number": 14 + } ] }, - "generated_at": "2026-09-02T01:06:14Z" + "generated_at": "2026-09-08T00:00:54Z" } diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 021886de..6c8e20b5 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -60,6 +60,15 @@ make test - Python 3.12+ - [uv](https://docs.astral.sh/uv/) — package and workspace manager - Node.js 18+ (only needed for the Prism mock server) +- Rust/Cargo (for `make factory-check`; the first run builds the reviewed `sf` revision) + +`make factory-check` and CI use `python3 scripts/sf.py`, which installs the +commit pinned in that launcher under `.local/software-factory/` without +replacing a global `sf`. It checks Cargo source provenance and the installed +binary's recorded SHA-256 before each invocation; an unrelated `sf` on `PATH` +is never used. A missing install needs network access; a damaged or unverifiable +install fails closed with its rebuild path. Do not remove rule documentation +to accommodate a different local tool version. ## Running the OpenAPI mock locally diff --git a/Makefile b/Makefile index 6d704179..e669eaed 100644 --- a/Makefile +++ b/Makefile @@ -152,9 +152,10 @@ check-docs: # runs a command, and without the flag `sf verify` scores it as fired on # the "commands are not enabled" finding instead of on its mutation -- # a rule proven by its own refusal to run. +.PHONY: factory-check factory-check: - sf verify --allow-commands - sf check --allow-commands + python3 scripts/sf.py verify --allow-commands + python3 scripts/sf.py check --allow-commands update-generated-lock: uv run python scripts/check_generated_lock.py --update diff --git a/artifacts/phase-gates/phase-15.15.json b/artifacts/phase-gates/phase-15.15.json index b692e143..a9dccbdb 100644 --- a/artifacts/phase-gates/phase-15.15.json +++ b/artifacts/phase-gates/phase-15.15.json @@ -7,12 +7,12 @@ "driver": "codex", "implementation": { "private_sha256": "27ad80932471b07200b9367b7aaead63837d9d02725105f4e1bf172841cbae7f", - "public_sha256": "6acfa322ca19032a3f29a5300301c2e0ffa1ad74cec3691d0157904860cd4bf0" + "public_sha256": "99215a360de9e32afd8a33338bc966ebd246ebec901c3005dce2b76efa9e14d2" }, "model": "gpt-5.4-mini", "report": "artifacts/phase-gates/reports/phase-15.15-isolated-runner-node.json", - "report_sha256": "dd7c5b4fe73648023fd542b79cd8adb49ccc18eca3eda8622e3fa21c5924befe", - "run_id": "20260907T012517-isolated_runner_node", + "report_sha256": "ecf736f4cffce4b50132ef608d8166588c01b02a67f4aad4b80fbc512c44d368", + "run_id": "20260907T234105-isolated_runner_node", "scenario": "isolated_runner_node", "status": "passed", "unsupported_assertions": [] diff --git a/artifacts/phase-gates/phase-16.1.json b/artifacts/phase-gates/phase-16.1.json index d92a7dcf..e295bd04 100644 --- a/artifacts/phase-gates/phase-16.1.json +++ b/artifacts/phase-gates/phase-16.1.json @@ -4,15 +4,15 @@ "runs": [ { "api_adapter": "local-devrig", - "driver": "codex", + "driver": "claude-code", "implementation": { "private_sha256": "430be1d01d7c99d0021ffecd42af771a99e4a67ac76b9aca8892294e517a8efc", - "public_sha256": "1d445b479d9912ef8022e7f17eb6ff1cc6c775e15c0a3f101dcf0b901c407961" + "public_sha256": "210dd03705f79936bc25d17e085166bb0bad303924dc3e11bb75f5a8885c364c" }, - "model": "gpt-5.4-mini", + "model": "claude-haiku-4-5", "report": "artifacts/phase-gates/reports/phase-16.1-eval-contract-reference-runner.json", - "report_sha256": "2ad3029ef87cd460b0df7af3dcf66c9fac023863c792bf3c61a285d851392652", - "run_id": "20260907T011240-eval_contract_reference_runner", + "report_sha256": "72cec14d236e18f2b649d4b3155bbb6b3e2a1183e63434e188e5797c3e7918d4", + "run_id": "20260907T225931-eval_contract_reference_runner", "scenario": "eval_contract_reference_runner", "status": "passed", "unsupported_assertions": [] diff --git a/artifacts/phase-gates/reports/phase-15.15-isolated-runner-node.json b/artifacts/phase-gates/reports/phase-15.15-isolated-runner-node.json index a9144ff9..8d55700a 100644 --- a/artifacts/phase-gates/reports/phase-15.15-isolated-runner-node.json +++ b/artifacts/phase-gates/reports/phase-15.15-isolated-runner-node.json @@ -1,5 +1,5 @@ { - "run_id": "20260907T012517-isolated_runner_node", + "run_id": "20260907T234105-isolated_runner_node", "scenario": "isolated_runner_node", "status": "passed", "api_adapter": "local-devrig", @@ -11,16 +11,16 @@ "sponsor": "codex", "auditor": "codex" }, - "started_at": "2026-09-07T01:25:17.372900Z", - "finished_at": "2026-09-07T01:27:56.552601Z", + "started_at": "2026-09-07T23:41:05.580828Z", + "finished_at": "2026-09-07T23:45:16.990820Z", "phase_results": [ { "phase_id": "prepare_runner_operator", "status": "completed", "assertion_results": [], - "started_at": "2026-09-07T01:25:18.167587Z", - "finished_at": "2026-09-07T01:25:20.059112Z", - "duration_seconds": 1.892 + "started_at": "2026-09-07T23:41:06.601187Z", + "finished_at": "2026-09-07T23:41:17.741211Z", + "duration_seconds": 11.14 }, { "phase_id": "node_operator_runner_flow", @@ -36,17 +36,17 @@ } } ], - "started_at": "2026-09-07T01:25:20.059128Z", - "finished_at": "2026-09-07T01:26:39.374115Z", - "duration_seconds": 79.314 + "started_at": "2026-09-07T23:41:17.741232Z", + "finished_at": "2026-09-07T23:43:32.525844Z", + "duration_seconds": 134.783 }, { "phase_id": "collect_runner_evidence", "status": "completed", "assertion_results": [], - "started_at": "2026-09-07T01:26:39.374219Z", - "finished_at": "2026-09-07T01:26:39.426477Z", - "duration_seconds": 0.052 + "started_at": "2026-09-07T23:43:32.525894Z", + "finished_at": "2026-09-07T23:43:32.562425Z", + "duration_seconds": 0.037 }, { "phase_id": "capture_runner_evidence", @@ -59,11 +59,11 @@ "evidence": { "runner_id": { "ok": true, - "value": "b85cc01f-d5b7-47e7-83a8-38d89e535036" + "value": "6ed734f3-4db9-4f93-b660-322e65ff6b1d" }, "runner_key_fingerprint": { "ok": true, - "value": "e2d1ecf70cd282c8" + "value": "a8dadf1fbb4050fa" }, "runner_import_root": { "ok": true, @@ -86,7 +86,7 @@ "evidence": { "job_id": { "ok": true, - "value": "7a4484cc-edfc-40a5-81e2-69d83227095c" + "value": "da2d7171-8c1e-4064-a22c-c14cfcf92afd" }, "terminal_status": { "ok": true, @@ -106,7 +106,7 @@ }, "lease_holder": { "ok": true, - "value": "b85cc01f-d5b7-47e7-83a8-38d89e535036" + "value": "6ed734f3-4db9-4f93-b660-322e65ff6b1d" } } }, @@ -117,11 +117,11 @@ "evidence": { "receipt_id": { "ok": true, - "value": "fc1851b8-0740-4306-a017-983f5efc09e5" + "value": "c0025bea-b7b1-4463-a84e-7377285490cc" }, "receipt_digest": { "ok": true, - "value": "fc67e0412570b2b76d808ac58ce47fab54f6fd24abfad26c748a4be4d87d4b24" + "value": "eb4fe58eb2151e882d2a546d5f36f1fecbd4dc9adb333d4d1c887249f7e2cbd0" }, "coordinator_accepted": { "ok": true, @@ -133,7 +133,7 @@ }, "published_at": { "ok": true, - "value": "2026-09-07T01:25:29.150515+00:00" + "value": "2026-09-07T23:41:37.304207+00:00" } } }, @@ -152,7 +152,7 @@ }, "signing_key_fingerprint": { "ok": true, - "value": "f971074c0db3052b" + "value": "c3cc1c658f1e22c9" }, "verify_exit_code": { "ok": true, @@ -324,8 +324,8 @@ } } ], - "started_at": "2026-09-07T01:26:39.426506Z", - "finished_at": "2026-09-07T01:26:39.429974Z", + "started_at": "2026-09-07T23:43:32.562445Z", + "finished_at": "2026-09-07T23:43:32.565211Z", "duration_seconds": 0.003 }, { @@ -341,9 +341,9 @@ } } ], - "started_at": "2026-09-07T01:26:39.429995Z", - "finished_at": "2026-09-07T01:27:56.552415Z", - "duration_seconds": 77.122 + "started_at": "2026-09-07T23:43:32.565231Z", + "finished_at": "2026-09-07T23:45:16.990455Z", + "duration_seconds": 104.424 } ], "assertion_results": [ @@ -363,11 +363,11 @@ "evidence": { "runner_id": { "ok": true, - "value": "b85cc01f-d5b7-47e7-83a8-38d89e535036" + "value": "6ed734f3-4db9-4f93-b660-322e65ff6b1d" }, "runner_key_fingerprint": { "ok": true, - "value": "e2d1ecf70cd282c8" + "value": "a8dadf1fbb4050fa" }, "runner_import_root": { "ok": true, @@ -390,7 +390,7 @@ "evidence": { "job_id": { "ok": true, - "value": "7a4484cc-edfc-40a5-81e2-69d83227095c" + "value": "da2d7171-8c1e-4064-a22c-c14cfcf92afd" }, "terminal_status": { "ok": true, @@ -410,7 +410,7 @@ }, "lease_holder": { "ok": true, - "value": "b85cc01f-d5b7-47e7-83a8-38d89e535036" + "value": "6ed734f3-4db9-4f93-b660-322e65ff6b1d" } } }, @@ -421,11 +421,11 @@ "evidence": { "receipt_id": { "ok": true, - "value": "fc1851b8-0740-4306-a017-983f5efc09e5" + "value": "c0025bea-b7b1-4463-a84e-7377285490cc" }, "receipt_digest": { "ok": true, - "value": "fc67e0412570b2b76d808ac58ce47fab54f6fd24abfad26c748a4be4d87d4b24" + "value": "eb4fe58eb2151e882d2a546d5f36f1fecbd4dc9adb333d4d1c887249f7e2cbd0" }, "coordinator_accepted": { "ok": true, @@ -437,7 +437,7 @@ }, "published_at": { "ok": true, - "value": "2026-09-07T01:25:29.150515+00:00" + "value": "2026-09-07T23:41:37.304207+00:00" } } }, @@ -456,7 +456,7 @@ }, "signing_key_fingerprint": { "ok": true, - "value": "f971074c0db3052b" + "value": "c3cc1c658f1e22c9" }, "verify_exit_code": { "ok": true, @@ -637,5 +637,5 @@ } ], "failure_message": null, - "artifact_root": "phase-15.15" + "artifact_root": "reseal-15.15" } \ No newline at end of file diff --git a/artifacts/phase-gates/reports/phase-16.1-eval-contract-reference-runner.json b/artifacts/phase-gates/reports/phase-16.1-eval-contract-reference-runner.json index bd6e0e16..7c27506c 100644 --- a/artifacts/phase-gates/reports/phase-16.1-eval-contract-reference-runner.json +++ b/artifacts/phase-gates/reports/phase-16.1-eval-contract-reference-runner.json @@ -1,22 +1,22 @@ { - "run_id": "20260907T011240-eval_contract_reference_runner", + "run_id": "20260907T225931-eval_contract_reference_runner", "scenario": "eval_contract_reference_runner", "status": "passed", "api_adapter": "local-devrig", "agent_drivers": { - "node_operator": "codex", - "auditor": "codex" + "node_operator": "claude-code", + "auditor": "claude-code" }, - "started_at": "2026-09-07T01:12:40.516697Z", - "finished_at": "2026-09-07T01:14:09.152373Z", + "started_at": "2026-09-07T22:59:31.017983Z", + "finished_at": "2026-09-07T23:00:55.423789Z", "phase_results": [ { "phase_id": "prepare_eval_operator_flow", "status": "completed", "assertion_results": [], - "started_at": "2026-09-07T01:12:40.881370Z", - "finished_at": "2026-09-07T01:12:41.953689Z", - "duration_seconds": 1.072 + "started_at": "2026-09-07T22:59:31.370957Z", + "finished_at": "2026-09-07T22:59:32.520151Z", + "duration_seconds": 1.149 }, { "phase_id": "node_operator_eval_flow", @@ -33,17 +33,17 @@ } } ], - "started_at": "2026-09-07T01:12:41.953709Z", - "finished_at": "2026-09-07T01:13:09.100411Z", - "duration_seconds": 27.146 + "started_at": "2026-09-07T22:59:32.520177Z", + "finished_at": "2026-09-07T23:00:07.125226Z", + "duration_seconds": 34.604 }, { "phase_id": "collect_eval_evidence", "status": "completed", "assertion_results": [], - "started_at": "2026-09-07T01:13:09.100498Z", - "finished_at": "2026-09-07T01:13:13.448766Z", - "duration_seconds": 4.348 + "started_at": "2026-09-07T23:00:07.125277Z", + "finished_at": "2026-09-07T23:00:11.056356Z", + "duration_seconds": 3.931 }, { "phase_id": "capture_eval_evidence", @@ -92,8 +92,8 @@ "eval_run_id": { "ok": true, "value": { - "run_one": "823057a30a814f6196a804afa5b903d4", - "run_two": "c8b05b5d2da747fe8da58b8f970c7841" + "run_one": "3709865ca9b54fcda9b9a6bab47e6746", + "run_two": "8cbb60e4f409467c873d84e8687c5871" } }, "terminal_status": { @@ -120,8 +120,8 @@ "runner_id": { "ok": true, "value": { - "run_one": "9d886cdd-ba8a-48d0-8b48-d981df0b14b4", - "run_two": "9d886cdd-ba8a-48d0-8b48-d981df0b14b4" + "run_one": "b2f0864d-42a1-4391-b0b8-858abd47cc89", + "run_two": "b2f0864d-42a1-4391-b0b8-858abd47cc89" } }, "receipt_digest": { @@ -341,8 +341,8 @@ } } ], - "started_at": "2026-09-07T01:13:13.448787Z", - "finished_at": "2026-09-07T01:13:13.474475Z", + "started_at": "2026-09-07T23:00:11.056380Z", + "finished_at": "2026-09-07T23:00:11.082064Z", "duration_seconds": 0.026 }, { @@ -358,9 +358,9 @@ } } ], - "started_at": "2026-09-07T01:13:13.474514Z", - "finished_at": "2026-09-07T01:14:09.152126Z", - "duration_seconds": 55.724 + "started_at": "2026-09-07T23:00:11.082102Z", + "finished_at": "2026-09-07T23:00:55.423636Z", + "duration_seconds": 44.34 } ], "assertion_results": [ @@ -417,8 +417,8 @@ "eval_run_id": { "ok": true, "value": { - "run_one": "823057a30a814f6196a804afa5b903d4", - "run_two": "c8b05b5d2da747fe8da58b8f970c7841" + "run_one": "3709865ca9b54fcda9b9a6bab47e6746", + "run_two": "8cbb60e4f409467c873d84e8687c5871" } }, "terminal_status": { @@ -445,8 +445,8 @@ "runner_id": { "ok": true, "value": { - "run_one": "9d886cdd-ba8a-48d0-8b48-d981df0b14b4", - "run_two": "9d886cdd-ba8a-48d0-8b48-d981df0b14b4" + "run_one": "b2f0864d-42a1-4391-b0b8-858abd47cc89", + "run_two": "b2f0864d-42a1-4391-b0b8-858abd47cc89" } }, "receipt_digest": { @@ -675,5 +675,5 @@ } ], "failure_message": null, - "artifact_root": "phase-16.1" + "artifact_root": "reseal-16.1" } \ No newline at end of file diff --git a/packages/cli/cli/_parser.py b/packages/cli/cli/_parser.py index fc4f5b45..a7a197f0 100644 --- a/packages/cli/cli/_parser.py +++ b/packages/cli/cli/_parser.py @@ -32,6 +32,7 @@ from cli.commands import ( credits as credits_mod, ) +from cli.commands import eval as eval_mod from cli.commands import ( referrals as referrals_mod, ) @@ -58,6 +59,7 @@ def build_parser() -> argparse.ArgumentParser: health.register(subparsers) doctor.register(subparsers) + eval_mod.register(subparsers) identity.register(subparsers) indexed.register(subparsers) listings.register(subparsers) diff --git a/packages/cli/cli/commands/eval/__init__.py b/packages/cli/cli/commands/eval/__init__.py new file mode 100644 index 00000000..47299d86 --- /dev/null +++ b/packages/cli/cli/commands/eval/__init__.py @@ -0,0 +1,6 @@ +# SPDX-License-Identifier: MIT +"""Portable eval contract commands.""" + +from cli.commands.eval.parser import register + +__all__ = ["register"] diff --git a/packages/cli/cli/commands/eval/_bundle.py b/packages/cli/cli/commands/eval/_bundle.py new file mode 100644 index 00000000..b15ecdbc --- /dev/null +++ b/packages/cli/cli/commands/eval/_bundle.py @@ -0,0 +1,153 @@ +# SPDX-License-Identifier: MIT +"""Handlers for portable eval creation and reproduction.""" + +from __future__ import annotations + +import argparse +import zipfile +from pathlib import Path + +from logion_eval_contract import ( + EvalContract, + EvalContractError, + contract_digest, + contract_to_json, + load_document, + pair_key, + parse_result_document, + result_digest, +) + +from cli._json import JsonObject +from cli._output import emit_json +from cli._version import __version__ as cli_version +from cli.commands.eval._scaffold import ( + _BUNDLE_MEDIA_TYPE, + _digest, + _fail, + _json_bytes, + _load_contract, + _safe_member, +) + +_HARNESS_ID = "logion-cli" +_MODEL_ID = "reference-subject" +_MODEL_VERSION = "1.0.0" +_EXIT_INVALID = 2 +_EXIT_ERROR = 1 +_EXIT_REFUSED = 3 + + +def _bundle_manifest( + contract: EvalContract, + subject: bytes, + fixture_entries: dict[str, JsonObject], + result_entries: list[JsonObject], +) -> JsonObject: + contract_bytes = _json_bytes(contract_to_json(contract)) + return { + "contract": { + "contract_digest": contract_digest(contract), + "digest": _digest(contract_bytes), + "path": "contract.json", + }, + "fixtures": fixture_entries, + "harness": {"id": _HARNESS_ID, "version": cli_version}, + "media_type": _BUNDLE_MEDIA_TYPE, + "results": result_entries, + "schema_version": 1, + "subject": {"digest": _digest(subject), "path": "subject.bin"}, + } + + +def _checked_results( + paths: list[str], contract: EvalContract, subject: bytes +) -> tuple[list[tuple[str, bytes]], list[JsonObject]]: + if len(paths) < 2: + raise ValueError("export requires at least two --result files") + files: list[tuple[str, bytes]] = [] + entries: list[JsonObject] = [] + parsed = [] + for index, path in enumerate(paths, start=1): + document, _ = load_document(path) + result = parse_result_document(document) + if result.contract_digest != contract_digest(contract): + raise ValueError(f"result {path!r} belongs to another contract") + if result.subject_digest != _digest(subject): + raise ValueError(f"result {path!r} belongs to another subject") + member = f"results/result-{index}.json" + raw = _json_bytes(result.to_json()) + files.append((member, raw)) + entries.append({ + "digest": _digest(raw), + "path": member, + "result_digest": result_digest(result), + }) + parsed.append(result) + if any(pair_key(item) != pair_key(parsed[0]) for item in parsed[1:]): + raise ValueError("results belong to different execution environments") + if ( + contract.determinism_class == "deterministic" + and len({result_digest(item) for item in parsed}) != 1 + ): + raise ValueError("deterministic run results do not match") + return files, entries + + +def _require_fixture_digest(raw: bytes, name: str, expected: str) -> None: + if _digest(raw) != expected: + raise ValueError(f"fixture {name!r} digest mismatch") + + +def handle_eval_export(args: argparse.Namespace) -> int: + """Package the contract, fixtures, subject, and run evidence.""" + output = Path(args.output) + try: + if output.exists() and not args.force: + raise FileExistsError( + f"{output} already exists; pass --force to replace it" + ) + contract = _load_contract(args.contract) + subject = Path(args.subject).read_bytes() + result_files, result_entries = _checked_results( + args.result, contract, subject + ) + base = Path(args.contract).resolve().parent + fixture_files: list[tuple[str, bytes]] = [] + fixture_entries: dict[str, JsonObject] = {} + for fixture in contract.fixtures: + name = _safe_member(fixture.name) + raw = (base / fixture.name).read_bytes() + _require_fixture_digest(raw, fixture.name, fixture.digest) + member = f"fixtures/{name}" + fixture_files.append((member, raw)) + fixture_entries[fixture.name] = { + "digest": fixture.digest, + "path": member, + } + manifest = _bundle_manifest( + contract, subject, fixture_entries, result_entries + ) + output.parent.mkdir(parents=True, exist_ok=True) + with zipfile.ZipFile(output, "w", zipfile.ZIP_DEFLATED) as archive: + archive.writestr("manifest.json", _json_bytes(manifest)) + archive.writestr( + "contract.json", _json_bytes(contract_to_json(contract)) + ) + archive.writestr("subject.bin", subject) + for member, raw in [*fixture_files, *result_files]: + archive.writestr(member, raw) + except EvalContractError as exc: + return _fail(exc.code, str(exc)) + except (OSError, ValueError, zipfile.BadZipFile) as exc: + return _fail("eval_bundle_invalid", str(exc)) + emit_json( + "logion.eval.export", + { + "bundle": str(output), + "bundle_digest": _digest(output.read_bytes()), + "contract_digest": contract_digest(contract), + "result_count": len(result_entries), + }, + ) + return 0 diff --git a/packages/cli/cli/commands/eval/_execution.py b/packages/cli/cli/commands/eval/_execution.py new file mode 100644 index 00000000..ca4a4157 --- /dev/null +++ b/packages/cli/cli/commands/eval/_execution.py @@ -0,0 +1,175 @@ +# SPDX-License-Identifier: MIT +"""Handlers for portable eval creation and reproduction.""" + +from __future__ import annotations + +import argparse +import json +from importlib.metadata import PackageNotFoundError, version +from pathlib import Path + +import logion_eval_contract as eval_contract_package +import yaml +from logion_eval_contract import ( + EvalContractError, + contract_digest, + parse_result_document, + result_digest, +) +from logion_runner.evals import ( + EvalExecutionError, + resolve_eval_job, +) + +from cli._output import emit_json +from cli.commands.eval._scaffold import ( + _digest, + _execute, + _fail, + _json_bytes, + _load_contract, + _scaffold_documents, + _write, +) + + +def _publish_result(args, contract, job, result): + """Route publishing through the public module for testable clients.""" + from cli.commands.eval import handlers + + return handlers._publish_result(args, contract, job, result) + + +_BUNDLE_MEDIA_TYPE = "application/vnd.logion.eval-reproduction.v1+zip" +_HARNESS_ID = "logion-cli" +_MODEL_ID = "reference-subject" +_MODEL_VERSION = "1.0.0" +_EXIT_INVALID = 2 +_EXIT_ERROR = 1 +_EXIT_REFUSED = 3 + + +def handle_eval_scaffold(args: argparse.Namespace) -> int: + """Create a complete starter contract and subject.""" + directory = Path(args.directory) + contract, subject = _scaffold_documents() + try: + _write( + directory / "eval-contract.yaml", + yaml.safe_dump(contract, sort_keys=False).encode(), + force=args.force, + ) + _write( + directory / "subject.json", + _json_bytes(subject), + force=args.force, + ) + parsed = _load_contract(str(directory / "eval-contract.yaml")) + except (EvalContractError, OSError) as exc: + return _fail("eval_scaffold_failed", str(exc)) + emit_json( + "logion.eval.scaffold", + { + "contract": str(directory / "eval-contract.yaml"), + "contract_digest": contract_digest(parsed), + "subject": str(directory / "subject.json"), + }, + ) + return 0 + + +def _validator_version() -> str: + try: + return version("logion-eval-contract") + except PackageNotFoundError: + return eval_contract_package.__version__ + + +def handle_eval_validate(args: argparse.Namespace) -> int: + """Validate a contract and resolve all declared fixtures.""" + try: + contract = _load_contract(args.contract) + base = Path(args.contract).resolve().parent + fixture_digests: dict[str, str] = {} + for fixture in contract.fixtures: + raw = (base / fixture.name).read_bytes() + actual = _digest(raw) + if actual != fixture.digest: + from logion_eval_contract import EvalFixtureDigestMismatch + + raise EvalFixtureDigestMismatch( + f"fixture {fixture.name!r} hashes to {actual}," + f" not {fixture.digest}" + ) + fixture_digests[fixture.name] = actual + except EvalContractError as exc: + return _fail(exc.code, str(exc)) + except OSError as exc: + return _fail("eval_fixture_unreadable", str(exc)) + emit_json( + "logion.eval.validate", + { + "contract_digest": contract_digest(contract), + "determinism_class": contract.determinism_class, + "fixture_digests": fixture_digests, + "valid": True, + "validator_import_root": ( + "site-packages" + if "site-packages" in str(eval_contract_package.__file__) + else "source-tree" + ), + "validator_package_version": _validator_version(), + }, + ) + return 0 + + +def handle_eval_run(args: argparse.Namespace) -> int: + """Resolve inputs, execute the subject, and emit a normalized result.""" + try: + contract = _load_contract(args.contract) + subject = Path(args.subject).read_bytes() + job = resolve_eval_job( + contract, + subject, + contract_dir=Path(args.contract).resolve().parent, + ) + outcome = _execute(contract, subject, args.contract) + published = ( + _publish_result(args, contract, job, outcome.result_document) + if args.publish + else None + ) + if args.output: + _write( + Path(args.output), + _json_bytes(outcome.result_document), + force=args.force, + ) + except EvalContractError as exc: + return _fail(exc.code, str(exc)) + except (OSError, FileExistsError, ValueError, json.JSONDecodeError) as exc: + return _fail("eval_input_unreadable", str(exc)) + except EvalExecutionError as exc: + return _fail("eval_execution_failed", str(exc), _EXIT_ERROR) + emit_json( + "logion.eval.run", + { + "assertion_outcomes": outcome.assertion_outcomes, + "executed": True, + "output": args.output, + "resolved": { + "contract_digest": job.contract_digest, + "evaluator_digest": contract.evaluator_requirement.digest, + "image": job.sandbox_profile["image"], + "sandbox_profile_digest": job.sandbox_profile_digest, + "subject_digest": job.subject_digest, + }, + "result": outcome.result_document, + "result_digest": result_digest( + parse_result_document(outcome.result_document) + ), + "published": published, + }, + ) + return 0 diff --git a/packages/cli/cli/commands/eval/_scaffold.py b/packages/cli/cli/commands/eval/_scaffold.py new file mode 100644 index 00000000..e1a01365 --- /dev/null +++ b/packages/cli/cli/commands/eval/_scaffold.py @@ -0,0 +1,186 @@ +# SPDX-License-Identifier: MIT +"""Handlers for portable eval creation and reproduction.""" + +from __future__ import annotations + +import hashlib +import json +from dataclasses import replace +from pathlib import Path, PurePosixPath + +from logion_eval_contract import ( + EvalContract, + contract_to_json, + parse_contract_file, +) +from logion_runner.evals import ( + execute_eval_contract, +) + +from cli._config import resolve_config_from_args +from cli._context import make_client +from cli._errors import emit_error_json +from cli._json import JsonObject, JsonValue +from cli._version import __version__ as cli_version + +_BUNDLE_MEDIA_TYPE = "application/vnd.logion.eval-reproduction.v1+zip" +_HARNESS_ID = "logion-cli" +_MODEL_ID = "reference-subject" +_MODEL_VERSION = "1.0.0" +_EXIT_INVALID = 2 +_EXIT_ERROR = 1 +_EXIT_REFUSED = 3 + + +def _safe_member(name: str) -> str: + """Reject absolute and parent-traversing bundle member names.""" + path = PurePosixPath(name) + if path.is_absolute() or ".." in path.parts or not path.parts: + raise ValueError(f"unsafe bundle member path: {name!r}") + return path.as_posix() + + +def _fail(code: str, message: str, exit_code: int = _EXIT_INVALID) -> int: + emit_error_json(code, message, exit_code) + return exit_code + + +def _digest(raw: bytes) -> str: + return hashlib.sha256(raw).hexdigest() + + +def _json_bytes(value: JsonValue) -> bytes: + return (json.dumps(value, indent=2, sort_keys=True) + "\n").encode() + + +def _write(path: Path, raw: bytes, *, force: bool) -> None: + if path.exists() and not force: + raise FileExistsError( + f"{path} already exists; pass --force to replace it" + ) + path.parent.mkdir(parents=True, exist_ok=True) + path.write_bytes(raw) + + +def _load_contract(path: str) -> EvalContract: + return parse_contract_file(path) + + +def _execute(contract: EvalContract, subject: bytes, contract_path: str): + return execute_eval_contract( + contract, + subject, + harness_id=_HARNESS_ID, + harness_version=cli_version, + model_id=_MODEL_ID, + model_version=_MODEL_VERSION, + contract_dir=Path(contract_path).resolve().parent, + ) + + +def _runner_credentials(path: str | None) -> tuple[str, str]: + if not path: + raise ValueError("--publish requires --runner-credentials") + payload = json.loads(Path(path).read_text(encoding="utf-8")) + if not isinstance(payload, dict): + raise TypeError("runner credentials must contain a JSON object") + runner_id = payload.get("runner_id") + runner_key = payload.get("runner_key") + if not isinstance(runner_id, str) or not runner_id: + raise ValueError("runner credentials are missing runner_id") + if not isinstance(runner_key, str) or not runner_key: + raise ValueError("runner credentials are missing runner_key") + return runner_id, runner_key + + +def _publish_result(args, contract, job, result) -> JsonObject: + """Publish through the handwritten SDK surfaces only.""" + runner_id, runner_key = _runner_credentials(args.runner_credentials) + config = resolve_config_from_args(args) + user_client = make_client(config) + runner_client = make_client(replace(config, api_key=runner_key)) + try: + uploaded = user_client.v1.evals.upload_contract( + contract_to_json(contract) + ) + remote_digest = uploaded.get("contract_digest") + if remote_digest != job.contract_digest: + raise ValueError("server and local contract digests differ") + user_client.v1.evals.validate_job( + job.contract_digest, + job.subject_digest, + {fixture.name: fixture.digest for fixture in contract.fixtures}, + ) + submitted_result = dict(result) + submitted_result.pop("contract_standing", None) + submitted = runner_client.v1.evals.submit_result(submitted_result) + finally: + runner_client.close() + user_client.close() + run_id = submitted.get("run_id") + if not isinstance(run_id, str) or not run_id: + raise ValueError("server returned no eval run id") + return { + "contract_digest": remote_digest, + "run_id": run_id, + "runner_id": runner_id, + "standing": uploaded.get("standing"), + "terminal_status": "succeeded", + } + + +def _scaffold_documents() -> tuple[JsonObject, JsonObject]: + subject: JsonObject = { + "input": { + "email": " CREATOR@EXAMPLE.COM ", + "name": " Example Creator ", + }, + "expected": { + "input": { + "email": "creator@example.com", + "name": "Example Creator", + } + }, + } + subject_digest = _digest(_json_bytes(subject)) + contract: JsonObject = { + "archetype": "exact_match", + "assertions": [ + { + "expected": 1, + "id": "output_matches_golden", + "metric": "cases_passed", + "operator": "eq", + } + ], + "budgets": [ + {"kind": "wall_seconds", "max_value": 60}, + {"kind": "output_bytes", "max_value": 1048576}, + ], + "determinism_class": "deterministic", + "evaluator_requirement": {"kind": "none"}, + "fixtures": [{"digest": subject_digest, "name": "subject.json"}], + "inputs": ["subject.json"], + "metrics": [ + { + "direction": "higher_is_better", + "id": "cases_passed", + "kind": "count", + } + ], + "outputs": [{"name": "result", "path": "outputs/result.json"}], + "redaction": {"fields": ["secret", "token"], "mode": "drop"}, + "runtime_requirements": [ + {"kind": "sandbox_profile", "value": "pinned-image"} + ], + "schema_version": 1, + "steps": [ + { + "action": "execute_subject", + "id": "run_subject", + "params": {"entrypoint": "normalize", "input": "subject.json"}, + } + ], + "subject": {"digest_constraint": "exact", "type": "agent_skill"}, + } + return contract, subject diff --git a/packages/cli/cli/commands/eval/_verify.py b/packages/cli/cli/commands/eval/_verify.py new file mode 100644 index 00000000..30a39698 --- /dev/null +++ b/packages/cli/cli/commands/eval/_verify.py @@ -0,0 +1,223 @@ +# SPDX-License-Identifier: MIT +"""Handlers for portable eval creation and reproduction.""" + +from __future__ import annotations + +import argparse +import json +import tempfile +import zipfile +from pathlib import Path + +from logion_eval_contract import ( + EvalContract, + EvalContractError, + contract_digest, + load_document, + parse_contract_document, + parse_result_document, + result_digest, +) +from logion_runner.evals import ( + EvalExecutionError, +) + +from cli._json import JsonObject +from cli._output import emit_json +from cli.commands.eval._bundle import _safe_member +from cli.commands.eval._scaffold import ( + _BUNDLE_MEDIA_TYPE, + _digest, + _execute, + _fail, + _load_contract, +) + +_HARNESS_ID = "logion-cli" +_MODEL_ID = "reference-subject" +_MODEL_VERSION = "1.0.0" +_EXIT_INVALID = 2 +_EXIT_ERROR = 1 +_EXIT_REFUSED = 3 + + +def _object(raw: bytes, where: str) -> JsonObject: + value = json.loads(raw) + if not isinstance(value, dict): + raise TypeError(f"{where} must contain a JSON object") + return value + + +def _entry(manifest: JsonObject, name: str) -> JsonObject: + value = manifest.get(name) + if not isinstance(value, dict): + raise TypeError(f"bundle manifest {name!r} must be an object") + return value + + +def _text(entry: JsonObject, name: str) -> str: + value = entry.get(name) + if not isinstance(value, str) or not value: + raise ValueError(f"bundle manifest field {name!r} must be text") + return value + + +def _read_member( + archive: zipfile.ZipFile, entry: JsonObject, label: str +) -> bytes: + path = _safe_member(_text(entry, "path")) + raw = archive.read(path) + if _digest(raw) != _text(entry, "digest"): + raise ValueError(f"bundle {label} digest mismatch") + return raw + + +def _open_bundle(path: str): + archive = zipfile.ZipFile(path) + manifest = _object(archive.read("manifest.json"), "manifest.json") + if manifest.get("schema_version") != 1: + raise ValueError("unsupported bundle schema_version") + if manifest.get("media_type") != _BUNDLE_MEDIA_TYPE: + raise ValueError("unsupported bundle media_type") + return archive, manifest + + +def _materialize_fixtures( + archive: zipfile.ZipFile, + manifest: JsonObject, + contract: EvalContract, + root: Path, +) -> None: + fixtures = manifest.get("fixtures") + if not isinstance(fixtures, dict): + raise TypeError("bundle manifest fixtures must be an object") + for fixture in contract.fixtures: + raw_entry = fixtures.get(fixture.name) + if not isinstance(raw_entry, dict): + raise TypeError(f"missing bundled fixture {fixture.name!r}") + raw = _read_member(archive, raw_entry, fixture.name) + target = root / _safe_member(fixture.name) + target.parent.mkdir(parents=True, exist_ok=True) + target.write_bytes(raw) + + +def _expected_result_digests( + archive: zipfile.ZipFile, manifest: JsonObject +) -> list[str]: + entries = manifest.get("results") + if not isinstance(entries, list) or len(entries) < 2: + raise ValueError("bundle must contain at least two results") + expected = [] + for index, raw_entry in enumerate(entries): + if not isinstance(raw_entry, dict): + raise TypeError("bundle result entry must be an object") + raw = _read_member(archive, raw_entry, f"result {index + 1}") + result = parse_result_document(_object(raw, "result")) + digest = result_digest(result) + if digest != _text(raw_entry, "result_digest"): + raise ValueError(f"bundle result {index + 1} digest mismatch") + expected.append(digest) + return expected + + +def _require_contract_digest( + contract: EvalContract, manifest_entry: JsonObject +) -> None: + if contract_digest(contract) != _text(manifest_entry, "contract_digest"): + raise ValueError("bundle contract semantic digest mismatch") + + +def handle_eval_verify(args: argparse.Namespace) -> int: + """Reproduce a bundle twice in a clean temporary workspace.""" + try: + archive, manifest = _open_bundle(args.bundle) + with archive, tempfile.TemporaryDirectory() as temp: + contract_raw = _read_member( + archive, _entry(manifest, "contract"), "contract" + ) + subject = _read_member( + archive, _entry(manifest, "subject"), "subject" + ) + root = Path(temp) + contract_path = root / "contract.json" + contract_path.write_bytes(contract_raw) + contract = _load_contract(str(contract_path)) + _require_contract_digest(contract, _entry(manifest, "contract")) + _materialize_fixtures(archive, manifest, contract, root) + expected = _expected_result_digests(archive, manifest) + first = _execute(contract, subject, str(contract_path)) + second = _execute(contract, subject, str(contract_path)) + actual = [ + result_digest(parse_result_document(outcome.result_document)) + for outcome in (first, second) + ] + if len(set(actual)) != 1 or set(actual) != set(expected): + return _fail( + "eval_reproduction_mismatch", + "clean-workspace runs do not match the exported results", + _EXIT_REFUSED, + ) + except EvalContractError as exc: + return _fail(exc.code, str(exc)) + except EvalExecutionError as exc: + return _fail("eval_execution_failed", str(exc), _EXIT_ERROR) + except ( + KeyError, + OSError, + TypeError, + ValueError, + json.JSONDecodeError, + zipfile.BadZipFile, + ) as exc: + return _fail("eval_bundle_invalid", str(exc)) + emit_json( + "logion.eval.verify", + { + "bundle": args.bundle, + "contract_digest": contract_digest(contract), + "reproduced": True, + "result_digest": actual[0], + "runs": 2, + }, + ) + return 0 + + +def handle_eval_inspect(args: argparse.Namespace) -> int: + """Inspect a result or bundle without executing it.""" + try: + if zipfile.is_zipfile(args.artifact): + archive, manifest = _open_bundle(args.artifact) + with archive: + contract_raw = _read_member( + archive, _entry(manifest, "contract"), "contract" + ) + contract = parse_contract_document( + _object(contract_raw, "contract.json") + ) + data: JsonObject = { + "artifact_type": "reproduction_bundle", + "contract_digest": contract_digest(contract), + "manifest": manifest, + } + else: + document, _ = load_document(args.artifact) + result = parse_result_document(document) + data = { + "artifact_type": "eval_result", + "result": result.to_json(), + "result_digest": result_digest(result), + } + except EvalContractError as exc: + return _fail("eval_result_invalid", str(exc)) + except ( + KeyError, + OSError, + TypeError, + ValueError, + json.JSONDecodeError, + zipfile.BadZipFile, + ) as exc: + return _fail("eval_artifact_invalid", str(exc)) + emit_json("logion.eval.inspect", data) + return 0 diff --git a/packages/cli/cli/commands/eval/handlers.py b/packages/cli/cli/commands/eval/handlers.py new file mode 100644 index 00000000..3d15a6d5 --- /dev/null +++ b/packages/cli/cli/commands/eval/handlers.py @@ -0,0 +1,33 @@ +# SPDX-License-Identifier: MIT +"""Public handlers for the eval command group.""" + +from cli.commands.eval import _scaffold +from cli.commands.eval._bundle import handle_eval_export +from cli.commands.eval._execution import ( + handle_eval_run, + handle_eval_scaffold, + handle_eval_validate, +) +from cli.commands.eval._verify import handle_eval_inspect, handle_eval_verify + +make_client = _scaffold.make_client + + +def _publish_result(args, contract, job, result): + """Keep the public compatibility seam for SDK-client injection tests.""" + original = _scaffold.make_client + _scaffold.make_client = make_client + try: + return _scaffold._publish_result(args, contract, job, result) + finally: + _scaffold.make_client = original + + +__all__ = [ + "handle_eval_export", + "handle_eval_inspect", + "handle_eval_run", + "handle_eval_scaffold", + "handle_eval_validate", + "handle_eval_verify", +] diff --git a/packages/cli/cli/commands/eval/parser.py b/packages/cli/cli/commands/eval/parser.py new file mode 100644 index 00000000..8c8ede65 --- /dev/null +++ b/packages/cli/cli/commands/eval/parser.py @@ -0,0 +1,96 @@ +# SPDX-License-Identifier: MIT +"""Parser registration for portable eval workflows.""" + +from __future__ import annotations + +import argparse + +from cli._options import COMMON_PARSER +from cli.commands.eval.handlers import ( + handle_eval_export, + handle_eval_inspect, + handle_eval_run, + handle_eval_scaffold, + handle_eval_validate, + handle_eval_verify, +) + + +def register(subparsers: argparse._SubParsersAction) -> None: + """Register the ``eval`` command tree.""" + parser = subparsers.add_parser( + "eval", help="Create, run, and reproduce portable evals" + ) + sub = parser.add_subparsers(dest="eval_command", required=True) + + scaffold = sub.add_parser( + "scaffold", help="Create a portable starter eval" + ) + scaffold.add_argument("directory", help="Directory to create") + scaffold.add_argument( + "--force", action="store_true", help="Replace scaffold files" + ) + scaffold.add_argument("--json", action="store_true", help="Emit JSON") + scaffold.set_defaults(handler=handle_eval_scaffold) + + validate = sub.add_parser( + "validate", + help="Validate a contract and its local fixtures", + parents=[COMMON_PARSER], + ) + validate.add_argument("contract", help="Contract file (YAML or JSON)") + validate.set_defaults(handler=handle_eval_validate) + + run = sub.add_parser( + "run", + help="Resolve and execute a local subject", + parents=[COMMON_PARSER], + ) + run.add_argument("contract", help="Contract file (YAML or JSON)") + run.add_argument("--subject", required=True, help="Subject file path") + run.add_argument("--output", help="Write the normalized result to FILE") + run.add_argument( + "--force", action="store_true", help="Replace the output file" + ) + run.add_argument( + "--publish", + action="store_true", + help="Upload the contract and submit the normalized result", + ) + run.add_argument( + "--runner-credentials", + help="JSON file containing runner_id and runner_key", + ) + run.set_defaults(handler=handle_eval_run) + + export = sub.add_parser( + "export", help="Package a contract and two run results" + ) + export.add_argument("contract", help="Contract file (YAML or JSON)") + export.add_argument("--subject", required=True, help="Subject file path") + export.add_argument( + "--result", + action="append", + required=True, + help="Normalized result file; pass at least twice", + ) + export.add_argument("--output", required=True, help="Bundle ZIP path") + export.add_argument( + "--force", action="store_true", help="Replace the output bundle" + ) + export.add_argument("--json", action="store_true", help="Emit JSON") + export.set_defaults(handler=handle_eval_export) + + verify = sub.add_parser( + "verify", help="Validate and reproduce an exported bundle twice" + ) + verify.add_argument("bundle", help="Reproduction bundle ZIP") + verify.add_argument("--json", action="store_true", help="Emit JSON") + verify.set_defaults(handler=handle_eval_verify) + + inspect = sub.add_parser( + "inspect", help="Inspect a normalized result or reproduction bundle" + ) + inspect.add_argument("artifact", help="Result JSON/YAML or bundle ZIP") + inspect.add_argument("--json", action="store_true", help="Emit JSON") + inspect.set_defaults(handler=handle_eval_inspect) diff --git a/packages/cli/pyproject.toml b/packages/cli/pyproject.toml index 3f23c4cc..280baf0a 100644 --- a/packages/cli/pyproject.toml +++ b/packages/cli/pyproject.toml @@ -20,6 +20,7 @@ classifiers = [ dependencies = [ "cryptography>=50.0.0,<51.0.0", "logion-client", + "logion-runner", "logion-skillmap", "pydantic>=2.7,<3.0.0", "pyyaml>=6.0,<7.0", @@ -38,6 +39,7 @@ recall = ["rapidfuzz>=3.9,<4.0"] [tool.uv.sources] logion-client = { workspace = true } +logion-runner = { workspace = true } logion-skillmap = { workspace = true } [project.scripts] diff --git a/packages/cli/tests/test_cli_eval.py b/packages/cli/tests/test_cli_eval.py new file mode 100644 index 00000000..d6db2a46 --- /dev/null +++ b/packages/cli/tests/test_cli_eval.py @@ -0,0 +1,269 @@ +# SPDX-License-Identifier: MIT +"""End-to-end tests for the public portable eval workflow.""" + +from __future__ import annotations + +import json +import zipfile +from pathlib import Path + +import pytest + +from cli._parser import build_parser +from cli.commands.eval import handlers + + +def _run(argv: list[str]) -> int: + args = build_parser().parse_args(argv) + return args.handler(args) + + +def _payload(capsys) -> dict: + return json.loads(capsys.readouterr().out) + + +def _creator_project(tmp_path: Path, capsys) -> tuple[Path, Path, Path, dict]: + project = tmp_path / "creator" + assert _run(["eval", "scaffold", str(project)]) == 0 + scaffold = _payload(capsys) + contract = project / "eval-contract.yaml" + subject = project / "subject.json" + assert scaffold["kind"] == "logion.eval.scaffold" + assert _run(["eval", "validate", str(contract)]) == 0 + validated = _payload(capsys) + assert validated["data"]["valid"] is True + return contract, subject, project, validated + + +def _run_results(contract: Path, subject: Path, project: Path, capsys): + results = [] + for index in (1, 2): + result = project / f"result-{index}.json" + assert ( + _run([ + "eval", + "run", + str(contract), + "--subject", + str(subject), + "--output", + str(result), + ]) + == 0 + ) + run = _payload(capsys) + assert run["kind"] == "logion.eval.run" + assert run["data"]["executed"] is True + results.append(result) + assert results[0].read_bytes() == results[1].read_bytes() + return results + + +def _export_bundle( + contract: Path, subject: Path, results, bundle: Path, capsys +): + assert ( + _run([ + "eval", + "export", + str(contract), + "--subject", + str(subject), + "--result", + str(results[0]), + "--result", + str(results[1]), + "--output", + str(bundle), + ]) + == 0 + ) + exported = _payload(capsys) + assert exported["data"]["result_count"] == 2 + + +def test_public_eval_creator_and_clean_consumer_flow( + tmp_path: Path, capsys +) -> None: + contract, subject, project, validated = _creator_project(tmp_path, capsys) + results = _run_results(contract, subject, project, capsys) + bundle = tmp_path / "reproduction.zip" + _export_bundle(contract, subject, results, bundle, capsys) + + assert _run(["eval", "verify", str(bundle)]) == 0 + verified = _payload(capsys) + assert verified["data"]["reproduced"] is True + assert verified["data"]["runs"] == 2 + + assert _run(["eval", "inspect", str(bundle)]) == 0 + inspected = _payload(capsys) + assert inspected["data"]["artifact_type"] == "reproduction_bundle" + assert ( + inspected["data"]["contract_digest"] + == validated["data"]["contract_digest"] + ) + + assert _run(["eval", "inspect", str(results[0])]) == 0 + assert _payload(capsys)["data"]["artifact_type"] == "eval_result" + + +def test_export_requires_two_run_results(tmp_path: Path, capsys) -> None: + project = tmp_path / "creator" + assert _run(["eval", "scaffold", str(project)]) == 0 + _payload(capsys) + contract = project / "eval-contract.yaml" + subject = project / "subject.json" + result = project / "result.json" + assert ( + _run([ + "eval", + "run", + str(contract), + "--subject", + str(subject), + "--output", + str(result), + ]) + == 0 + ) + _payload(capsys) + + assert ( + _run([ + "eval", + "export", + str(contract), + "--subject", + str(subject), + "--result", + str(result), + "--output", + str(tmp_path / "bundle.zip"), + ]) + == 2 + ) + error = json.loads(capsys.readouterr().err) + assert error["kind"] == "logion.error" + assert error["data"]["code"] == "eval_bundle_invalid" + + +def test_eval_run_publish_uses_user_and_runner_sdk_clients( + tmp_path: Path, capsys, monkeypatch: pytest.MonkeyPatch +) -> None: + project = tmp_path / "creator" + assert _run(["eval", "scaffold", str(project)]) == 0 + scaffold = _payload(capsys) + credentials = tmp_path / "runner.json" + credentials.write_text( + json.dumps({"runner_id": "runner-1", "runner_key": "runner-key"}) + ) + + class Evals: + def __init__(self, runner: bool = False) -> None: + self.runner = runner + + def upload_contract(self, document): + assert not self.runner + assert document + return { + "contract_digest": scaffold["data"]["contract_digest"], + "standing": "unreviewed", + } + + def validate_job(self, contract_ref, subject_digest, fixture_digests): + assert not self.runner + assert contract_ref == scaffold["data"]["contract_digest"] + assert subject_digest + assert set(fixture_digests) == {"subject.json"} + return {"valid": True} + + def submit_result(self, result): + assert self.runner + assert "contract_standing" not in result + return {"run_id": "run-1"} + + class Client: + def __init__(self, runner: bool = False) -> None: + self.v1 = type("V1", (), {"evals": Evals(runner)})() + + def close(self) -> None: + pass + + clients = iter((Client(), Client(runner=True))) + monkeypatch.setattr(handlers, "make_client", lambda _config: next(clients)) + result = project / "result.json" + + assert ( + _run([ + "eval", + "run", + str(project / "eval-contract.yaml"), + "--subject", + str(project / "subject.json"), + "--output", + str(result), + "--publish", + "--runner-credentials", + str(credentials), + "--base-url", + "http://example.test", + ]) + == 0 + ) + payload = _payload(capsys) + assert payload["data"]["published"] == { + "contract_digest": scaffold["data"]["contract_digest"], + "run_id": "run-1", + "runner_id": "runner-1", + "standing": "unreviewed", + "terminal_status": "succeeded", + } + + +def test_verify_rejects_tampered_bundle(tmp_path: Path, capsys) -> None: + bad = tmp_path / "bad.zip" + with zipfile.ZipFile(bad, "w") as archive: + archive.writestr( + "manifest.json", + json.dumps({"schema_version": 1, "media_type": "wrong"}), + ) + assert _run(["eval", "verify", str(bad)]) == 2 + error = json.loads(capsys.readouterr().err) + assert error["data"]["code"] == "eval_bundle_invalid" + + +def test_eval_help_exposes_complete_workflow(capsys) -> None: + parser = build_parser() + with pytest.raises(SystemExit) as caught: + parser.parse_args(["eval", "--help"]) + assert caught.value.code == 0 + output = capsys.readouterr().out + for command in ( + "scaffold", + "validate", + "run", + "export", + "verify", + "inspect", + ): + assert command in output + + +def test_validate_without_installed_distribution_metadata( + tmp_path, capsys, monkeypatch +): + from importlib.metadata import PackageNotFoundError + + from cli.commands.eval import _execution + + def unavailable(name): + raise PackageNotFoundError(name) + + monkeypatch.setattr(_execution, "version", unavailable) + project = tmp_path / "creator" + assert _run(["eval", "scaffold", str(project)]) == 0 + _payload(capsys) + assert _run(["eval", "validate", str(project / "eval-contract.yaml")]) == 0 + assert _payload(capsys)["data"]["validator_package_version"] == ( + _execution.eval_contract_package.__version__ + ) diff --git a/packages/client/src/logion/v1/__init__.py b/packages/client/src/logion/v1/__init__.py index 44898ac1..80c8eb11 100644 --- a/packages/client/src/logion/v1/__init__.py +++ b/packages/client/src/logion/v1/__init__.py @@ -21,7 +21,29 @@ from logion.v1._resources.reports import ReportsResource from logion.v1._resources.resource_feedback import ResourceFeedbackResource from logion.v1._resources.resources import ResourcesResource +from logion.v1._resources.runners import RunnersResource from logion.v1._resources.usage_receipts import UsageReceiptResource +from logion.v1.eval_types import ( + EvalErrorResponse as EvalErrorResponse, +) +from logion.v1.eval_types import ( + SubmitEvalResultRequest as SubmitEvalResultRequest, +) +from logion.v1.eval_types import ( + SubmitEvalResultResponse as SubmitEvalResultResponse, +) +from logion.v1.eval_types import ( + UploadEvalContractRequest as UploadEvalContractRequest, +) +from logion.v1.eval_types import ( + UploadEvalContractResponse as UploadEvalContractResponse, +) +from logion.v1.eval_types import ( + ValidateEvalJobRequest as ValidateEvalJobRequest, +) +from logion.v1.eval_types import ( + ValidateEvalJobResponse as ValidateEvalJobResponse, +) class V1Namespace: @@ -43,6 +65,7 @@ def __init__(self, http: HttpClient) -> None: self.referrals = ReferralsResource(http) self.github_setup = GithubSetupResource(http) self.resources = ResourcesResource(http) + self.runners = RunnersResource(http) self.resource_feedback = ResourceFeedbackResource(http) self.usage_receipts = UsageReceiptResource(http) self.evals = EvalsResource(http) diff --git a/packages/client/src/logion/v1/_operation_map.py b/packages/client/src/logion/v1/_operation_map.py index e30b5559..bf223f00 100644 --- a/packages/client/src/logion/v1/_operation_map.py +++ b/packages/client/src/logion/v1/_operation_map.py @@ -39,6 +39,7 @@ "get_eval_contract": "client.v1.evals.get_contract", "validate_eval_job": "client.v1.evals.validate_job", "submit_eval_result": "client.v1.evals.submit_result", + "enroll_runner": "client.v1.runners.enroll", # Courses "create_course": "client.v1.courses.create", "get_course": "client.v1.courses.get", @@ -144,7 +145,6 @@ UNSUPPORTED_OPERATIONS: dict[str, str] = { # Runner/coordinator endpoints are generated for contract compatibility, # but have no stable handwritten SDK resource surface yet. - "enroll_runner": ("Runner operator endpoint; no stable SDK resource yet."), "rotate_runner_key": ( "Runner operator endpoint; no stable SDK resource yet." ), diff --git a/packages/client/src/logion/v1/_resources/runners.py b/packages/client/src/logion/v1/_resources/runners.py new file mode 100644 index 00000000..920fdeed --- /dev/null +++ b/packages/client/src/logion/v1/_resources/runners.py @@ -0,0 +1,23 @@ +# SPDX-License-Identifier: MIT +"""Runner operator resource.""" + +from __future__ import annotations + +from logion._http import HttpClient +from logion._json import JsonObject + + +class RunnersResource: + """Provision credentials for reference runners.""" + + def __init__(self, http: HttpClient) -> None: + self._http = http + + def enroll(self, name: str) -> JsonObject: + """Enroll one runner and return its one-time credentials.""" + name = name.strip() + if not name: + raise ValueError("runner name must not be empty") + return self._http.request_object( + "POST", "/v1/runners/enroll", json={"name": name} + ) diff --git a/packages/client/src/logion/v1/eval_types.py b/packages/client/src/logion/v1/eval_types.py new file mode 100644 index 00000000..b491ce50 --- /dev/null +++ b/packages/client/src/logion/v1/eval_types.py @@ -0,0 +1,22 @@ +# SPDX-License-Identifier: MIT +"""Public SDK types for eval API operations.""" + +from logion.v1._types.generated.v1 import ( + EvalErrorResponse, + SubmitEvalResultRequest, + SubmitEvalResultResponse, + UploadEvalContractRequest, + UploadEvalContractResponse, + ValidateEvalJobRequest, + ValidateEvalJobResponse, +) + +__all__ = [ + "EvalErrorResponse", + "SubmitEvalResultRequest", + "SubmitEvalResultResponse", + "UploadEvalContractRequest", + "UploadEvalContractResponse", + "ValidateEvalJobRequest", + "ValidateEvalJobResponse", +] diff --git a/packages/client/tests/test_v1_eval_types.py b/packages/client/tests/test_v1_eval_types.py new file mode 100644 index 00000000..7b082dd0 --- /dev/null +++ b/packages/client/tests/test_v1_eval_types.py @@ -0,0 +1,38 @@ +# SPDX-License-Identifier: MIT +"""Public eval SDK type exports.""" + +from logion.v1 import ( + EvalErrorResponse, + SubmitEvalResultRequest, + SubmitEvalResultResponse, + UploadEvalContractRequest, + UploadEvalContractResponse, + ValidateEvalJobRequest, + ValidateEvalJobResponse, +) + + +def test_eval_api_types_are_public() -> None: + assert UploadEvalContractRequest(document={}).document == {} + assert UploadEvalContractResponse( + contract_digest="a" * 64, + created=True, + media_type="application/json", + ).created + assert ( + ValidateEvalJobRequest( + contract_ref="contract", + subject_digest="b" * 64, + fixture_digests={}, + ).contract_ref + == "contract" + ) + assert ValidateEvalJobResponse(contract_digest="a" * 64, valid=True).valid + assert SubmitEvalResultRequest(result={}).result == {} + assert ( + SubmitEvalResultResponse( + created=True, result_digest="c" * 64, run_id="run" + ).run_id + == "run" + ) + assert EvalErrorResponse(detail="invalid").detail == "invalid" diff --git a/packages/client/tests/test_v1_runners_resource.py b/packages/client/tests/test_v1_runners_resource.py new file mode 100644 index 00000000..da5d2f21 --- /dev/null +++ b/packages/client/tests/test_v1_runners_resource.py @@ -0,0 +1,37 @@ +"""Tests for the handwritten runner resource.""" + +from unittest.mock import MagicMock + +import pytest + +from logion._http import HttpClient +from logion.v1._resources.runners import RunnersResource + + +def test_enroll_runner_uses_public_endpoint() -> None: + http = MagicMock(spec=HttpClient) + http.request_object.return_value = {"runner_id": "runner-1"} + + result = RunnersResource(http).enroll("creator-runner") + + assert result == {"runner_id": "runner-1"} + http.request_object.assert_called_once_with( + "POST", "/v1/runners/enroll", json={"name": "creator-runner"} + ) + + +def test_enroll_runner_rejects_empty_name() -> None: + http = MagicMock(spec=HttpClient) + + with pytest.raises(ValueError, match="must not be empty"): + RunnersResource(http).enroll(" ") + + http.request_object.assert_not_called() + + +def test_enroll_strips_surrounding_whitespace() -> None: + http = MagicMock(spec=HttpClient) + RunnersResource(http).enroll(" creator-runner\t") + http.request_object.assert_called_once_with( + "POST", "/v1/runners/enroll", json={"name": "creator-runner"} + ) diff --git a/packages/landing/landing/content/docs.json b/packages/landing/landing/content/docs.json index 2407537e..633abd3d 100644 --- a/packages/landing/landing/content/docs.json +++ b/packages/landing/landing/content/docs.json @@ -260,6 +260,13 @@ "summary": "1 CLI command.", "title": "logion doctor" }, + "cli/eval": { + "body": "# logion eval\n\n6 commands in this group.\n\n- [`logion eval export`](#logion-eval-export)\n- [`logion eval inspect`](#logion-eval-inspect)\n- [`logion eval run`](#logion-eval-run)\n- [`logion eval scaffold`](#logion-eval-scaffold)\n- [`logion eval validate`](#logion-eval-validate)\n- [`logion eval verify`](#logion-eval-verify)\n\n## logion eval export\n\n```bash\nlogion eval export CONTRACT --subject SUBJECT --result RESULT --output OUTPUT [--force] [--json]\n```\n\n| Option | Value | Default | Description |\n| --- | --- | --- | --- |\n| `CONTRACT` | string | — | Contract file (YAML or JSON) |\n| `--subject` | string | — | Subject file path |\n| `--result` | string | — | Normalized result file; pass at least twice |\n| `--output` | string | — | Bundle ZIP path |\n| `--force` | flag | — | Replace the output bundle |\n| `--json` | flag | — | Emit JSON |\n\n## logion eval inspect\n\n```bash\nlogion eval inspect ARTIFACT [--json]\n```\n\n| Option | Value | Default | Description |\n| --- | --- | --- | --- |\n| `ARTIFACT` | string | — | Result JSON/YAML or bundle ZIP |\n| `--json` | flag | — | Emit JSON |\n\n## logion eval run\n\n```bash\nlogion eval run [--api-key API_KEY] [--base-url BASE_URL] [--json] [--timeout TIMEOUT] [--max-retries MAX_RETRIES] [--no-onboarding] CONTRACT --subject SUBJECT [--output OUTPUT] [--force] [--publish] [--runner-credentials RUNNER_CREDENTIALS]\n```\n\n| Option | Value | Default | Description |\n| --- | --- | --- | --- |\n| `--api-key` | string | — | |\n| `--base-url` | string | — | |\n| `--json` | flag | — | |\n| `--timeout` | float | — | |\n| `--max-retries` | int | — | |\n| `--no-onboarding` | flag | — | Never run first-run onboarding for this invocation. |\n| `CONTRACT` | string | — | Contract file (YAML or JSON) |\n| `--subject` | string | — | Subject file path |\n| `--output` | string | — | Write the normalized result to FILE |\n| `--force` | flag | — | Replace the output file |\n| `--publish` | flag | — | Upload the contract and submit the normalized result |\n| `--runner-credentials` | string | — | JSON file containing runner_id and runner_key |\n\n## logion eval scaffold\n\n```bash\nlogion eval scaffold DIRECTORY [--force] [--json]\n```\n\n| Option | Value | Default | Description |\n| --- | --- | --- | --- |\n| `DIRECTORY` | string | — | Directory to create |\n| `--force` | flag | — | Replace scaffold files |\n| `--json` | flag | — | Emit JSON |\n\n## logion eval validate\n\n```bash\nlogion eval validate [--api-key API_KEY] [--base-url BASE_URL] [--json] [--timeout TIMEOUT] [--max-retries MAX_RETRIES] [--no-onboarding] CONTRACT\n```\n\n| Option | Value | Default | Description |\n| --- | --- | --- | --- |\n| `--api-key` | string | — | |\n| `--base-url` | string | — | |\n| `--json` | flag | — | |\n| `--timeout` | float | — | |\n| `--max-retries` | int | — | |\n| `--no-onboarding` | flag | — | Never run first-run onboarding for this invocation. |\n| `CONTRACT` | string | — | Contract file (YAML or JSON) |\n\n## logion eval verify\n\n```bash\nlogion eval verify BUNDLE [--json]\n```\n\n| Option | Value | Default | Description |\n| --- | --- | --- | --- |\n| `BUNDLE` | string | — | Reproduction bundle ZIP |\n| `--json` | flag | — | Emit JSON |\n", + "kind": "cli", + "slug": "cli/eval", + "summary": "6 CLI commands.", + "title": "logion eval" + }, "cli/feedback": { "body": "# logion feedback\n\n3 commands in this group.\n\n- [`logion feedback list`](#logion-feedback-list)\n- [`logion feedback submit`](#logion-feedback-submit)\n- [`logion feedback summary`](#logion-feedback-summary)\n\n## logion feedback list\n\n```bash\nlogion feedback list [--api-key API_KEY] [--base-url BASE_URL] [--json] [--timeout TIMEOUT] [--max-retries MAX_RETRIES] [--no-onboarding] [--mine]\n```\n\n| Option | Value | Default | Description |\n| --- | --- | --- | --- |\n| `--api-key` | string | — | |\n| `--base-url` | string | — | |\n| `--json` | flag | — | |\n| `--timeout` | float | — | |\n| `--max-retries` | int | — | |\n| `--no-onboarding` | flag | — | Never run first-run onboarding for this invocation. |\n| `--mine` | flag | `True` | |\n\n**Calls**\n\n- [`list_my_feedback`](/docs/api/resource-feedback#list-the-current-agent-s-feedback)\n\n## logion feedback submit\n\n```bash\nlogion feedback submit [--api-key API_KEY] [--base-url BASE_URL] [--json] [--timeout TIMEOUT] [--max-retries MAX_RETRIES] [--no-onboarding] RESOURCE_ID VERSION_ID --rating RATING [--usefulness USEFULNESS] [--reliability RELIABILITY] [--tool-safety TOOL_SAFETY] [--acquisition-channel ACQUISITION_CHANNEL] [--force] [--source-receipt-id SOURCE_RECEIPT_ID] [--token-efficiency TOKEN_EFFICIENCY] [--completed-task] [--not-completed-task] --task-class TASK_CLASS [--body BODY]\n```\n\n| Option | Value | Default | Description |\n| --- | --- | --- | --- |\n| `--api-key` | string | — | |\n| `--base-url` | string | — | |\n| `--json` | flag | — | |\n| `--timeout` | float | — | |\n| `--max-retries` | int | — | |\n| `--no-onboarding` | flag | — | Never run first-run onboarding for this invocation. |\n| `RESOURCE_ID` | string | — | |\n| `VERSION_ID` | string | — | |\n| `--rating` | 1 \\| 2 \\| 3 \\| 4 \\| 5 | — | |\n| `--usefulness` | int | — | |\n| `--reliability` | int | — | |\n| `--tool-safety` | int | — | |\n| `--acquisition-channel` | string | — | Override the channel; resolved from the local acquisition receipt when omitted. |\n| `--force` | flag | — | Revise feedback already submitted for this version/task class. |\n| `--source-receipt-id` | string | — | |\n| `--token-efficiency` | int | — | |\n| `--completed-task` | flag | — | |\n| `--not-completed-task` | flag | `True` | |\n| `--task-class` | string | — | |\n| `--body` | string | — | |\n\n**Calls**\n\n- [`submit_feedback`](/docs/api/resource-feedback#submit-feedback-for-a-resource-version)\n\n## logion feedback summary\n\n```bash\nlogion feedback summary [--api-key API_KEY] [--base-url BASE_URL] [--json] [--timeout TIMEOUT] [--max-retries MAX_RETRIES] [--no-onboarding] RESOURCE_ID\n```\n\n| Option | Value | Default | Description |\n| --- | --- | --- | --- |\n| `--api-key` | string | — | |\n| `--base-url` | string | — | |\n| `--json` | flag | — | |\n| `--timeout` | float | — | |\n| `--max-retries` | int | — | |\n| `--no-onboarding` | flag | — | Never run first-run onboarding for this invocation. |\n| `RESOURCE_ID` | string | — | |\n\n**Calls**\n\n- [`get_feedback_summary`](/docs/api/resource-feedback#get-aggregate-feedback-summary-for-a-resource)\n", "kind": "cli", @@ -317,10 +324,10 @@ "title": "logion onboarding" }, "cli/overview": { - "body": "# CLI Reference\n\nEvery command the `logion` binary accepts: **24 groups**, generated by walking the same argparse tree the binary builds at runtime.\n\nThe CLI is the execution layer. It sends the API key, retries, and renders JSON, so an agent should reach for a command before an HTTP call.\n\n```bash\ncurl -fsSL https://logion.sh/install.sh | sh\nlogion --help\n```\n\n**66 of 129 API operations** are reachable from a command, and each one links to the other. The mapping is derived from the code, not maintained by hand — see [How these docs stay true](/docs/reference/how-these-docs-are-built).\n\n## Groups\n\n| Group | Commands |\n| --- | --- |\n| [logion admin](/docs/cli/admin) | 13 CLI commands. |\n| [logion bounties](/docs/cli/bounties) | 19 CLI commands. |\n| [logion completion](/docs/cli/completion) | 1 CLI command. |\n| [logion course-reviews](/docs/cli/course-reviews) | 5 CLI commands. |\n| [logion courses](/docs/cli/courses) | 26 CLI commands. |\n| [logion credits](/docs/cli/credits) | 5 CLI commands. |\n| [logion docs](/docs/cli/docs) | 1 CLI command. |\n| [logion doctor](/docs/cli/doctor) | 1 CLI command. |\n| [logion feedback](/docs/cli/feedback) | 3 CLI commands. |\n| [logion health](/docs/cli/health) | 1 CLI command. |\n| [logion identity](/docs/cli/identity) | 7 CLI commands. |\n| [logion indexed](/docs/cli/indexed) | 1 CLI command. |\n| [logion integrations](/docs/cli/integrations) | 4 CLI commands. |\n| [logion listings](/docs/cli/listings) | 1 CLI command. |\n| [logion notifications](/docs/cli/notifications) | 3 CLI commands. |\n| [logion onboarding](/docs/cli/onboarding) | 1 CLI command. |\n| [logion payments](/docs/cli/payments) | 5 CLI commands. |\n| [logion recall](/docs/cli/recall) | 2 CLI commands. |\n| [logion referrals](/docs/cli/referrals) | 4 CLI commands. |\n| [logion reports](/docs/cli/reports) | 1 CLI command. |\n| [logion resources](/docs/cli/resources) | 8 CLI commands. |\n| [logion skills](/docs/cli/skills) | 11 CLI commands. |\n| [logion update](/docs/cli/update) | 1 CLI command. |\n| [logion usage](/docs/cli/usage) | 4 CLI commands. |\n", + "body": "# CLI Reference\n\nEvery command the `logion` binary accepts: **25 groups**, generated by walking the same argparse tree the binary builds at runtime.\n\nThe CLI is the execution layer. It sends the API key, retries, and renders JSON, so an agent should reach for a command before an HTTP call.\n\n```bash\ncurl -fsSL https://logion.sh/install.sh | sh\nlogion --help\n```\n\n**66 of 129 API operations** are reachable from a command, and each one links to the other. The mapping is derived from the code, not maintained by hand — see [How these docs stay true](/docs/reference/how-these-docs-are-built).\n\n## Groups\n\n| Group | Commands |\n| --- | --- |\n| [logion admin](/docs/cli/admin) | 13 CLI commands. |\n| [logion bounties](/docs/cli/bounties) | 19 CLI commands. |\n| [logion completion](/docs/cli/completion) | 1 CLI command. |\n| [logion course-reviews](/docs/cli/course-reviews) | 5 CLI commands. |\n| [logion courses](/docs/cli/courses) | 26 CLI commands. |\n| [logion credits](/docs/cli/credits) | 5 CLI commands. |\n| [logion docs](/docs/cli/docs) | 1 CLI command. |\n| [logion doctor](/docs/cli/doctor) | 1 CLI command. |\n| [logion eval](/docs/cli/eval) | 6 CLI commands. |\n| [logion feedback](/docs/cli/feedback) | 3 CLI commands. |\n| [logion health](/docs/cli/health) | 1 CLI command. |\n| [logion identity](/docs/cli/identity) | 7 CLI commands. |\n| [logion indexed](/docs/cli/indexed) | 1 CLI command. |\n| [logion integrations](/docs/cli/integrations) | 4 CLI commands. |\n| [logion listings](/docs/cli/listings) | 1 CLI command. |\n| [logion notifications](/docs/cli/notifications) | 3 CLI commands. |\n| [logion onboarding](/docs/cli/onboarding) | 1 CLI command. |\n| [logion payments](/docs/cli/payments) | 5 CLI commands. |\n| [logion recall](/docs/cli/recall) | 2 CLI commands. |\n| [logion referrals](/docs/cli/referrals) | 4 CLI commands. |\n| [logion reports](/docs/cli/reports) | 1 CLI command. |\n| [logion resources](/docs/cli/resources) | 8 CLI commands. |\n| [logion skills](/docs/cli/skills) | 11 CLI commands. |\n| [logion update](/docs/cli/update) | 1 CLI command. |\n| [logion usage](/docs/cli/usage) | 4 CLI commands. |\n", "kind": "cli", "slug": "cli/overview", - "summary": "24 command groups, generated from the argparse tree.", + "summary": "25 command groups, generated from the argparse tree.", "title": "CLI Reference" }, "cli/payments": { @@ -443,7 +450,7 @@ "title": "Use Observation and Feedback" }, "index": { - "body": "# Logion documentation\n\nLogion measures whether the skills, plugins, MCP servers and models an agent installs actually do what they claim, and publishes the method so anyone can reproduce it.\n\nTwo references and a set of guides. The references are generated from the contract and the CLI itself, so they cannot drift — see [How these docs are built](/docs/reference/how-these-docs-are-built).\n\n## Guides\n\nWritten by hand. Start here.\n\n- [Bounties and Referrals](/docs/guides/bounties-and-referrals) — Understand credit-funded bounties and referral rewards.\n- [Core Concepts](/docs/guides/concepts) — Understand courses, versions, capabilities, entitlements, and roles.\n- [Creating Courses](/docs/guides/creating-courses) — Author, declare, upload, review, and publish a course safely.\n- [Credits and Purchases](/docs/guides/credits-and-purchases) — Learn how credits, free courses, paid purchases, and approvals work.\n- [Getting Started](/docs/guides/getting-started) — Install, authenticate, discover, acquire, and use your first course.\n- [Marketplace Loop](/docs/guides/marketplace-loop) — The complete Logion marketplace loop: search, inspect, acquire, install, review, and bounty.\n- [Course Reviews](/docs/guides/reviews) — File honest agent reviews automatically after meaningful course use.\n- [Safety and Trust](/docs/guides/safety) — Inspect permissions, understand review limits, and report unsafe content.\n- [Use Observation and Feedback](/docs/guides/use-observation-and-feedback) — Observe use of natively installed resources and report honest feedback under an explicit consent mode.\n\n## API Reference\n\n129 operations across 28 groups, generated from the OpenAPI contract.\n\n- [Overview](/docs/api/overview) — 129 operations across 28 groups, generated from the OpenAPI contract.\n- [Admin · Agents](/docs/api/admin-agents) — 3 API operations tagged `admin-agents`.\n- [Admin · Bounties](/docs/api/admin-bounties) — 4 API operations tagged `admin-bounties`.\n- [Admin · Courses](/docs/api/admin-courses) — 3 API operations tagged `admin-courses`.\n- [Admin · Indexing](/docs/api/admin-indexing) — 8 API operations tagged `admin-indexing`.\n- [Admin · Referrals](/docs/api/admin-referrals) — 1 API operation tagged `admin-referrals`.\n- [Admin · Reports](/docs/api/admin-reports) — 4 API operations tagged `admin-reports`.\n- [Admin · Users](/docs/api/admin-users) — 3 API operations tagged `admin-users`.\n- [AI Catalog](/docs/api/ai-catalog) — 1 API operation tagged `ai-catalog`.\n- [ARD Discovery](/docs/api/ard) — 5 API operations tagged `ard`.\n- [Bounties](/docs/api/bounties) — 15 API operations tagged `bounties`.\n- [Capabilities](/docs/api/capabilities) — 1 API operation tagged `capabilities`.\n- [Course Reviews](/docs/api/course-reviews) — 7 API operations tagged `course-reviews`.\n- [Courses](/docs/api/courses) — 15 API operations tagged `courses`.\n- [Credits](/docs/api/credits) — 4 API operations tagged `credits`.\n- [Evals](/docs/api/evals) — 4 API operations tagged `evals`.\n- [Executions](/docs/api/executions) — 4 API operations tagged `executions`.\n- [General](/docs/api/general) — 2 API operations tagged `general`.\n- [Identity](/docs/api/identity) — 9 API operations tagged `identity`.\n- [Indexed Listings](/docs/api/indexed-listings) — 1 API operation tagged `indexed-listings`.\n- [Listings](/docs/api/listings) — 1 API operation tagged `listings`.\n- [Notifications](/docs/api/notifications) — 2 API operations tagged `notifications`.\n- [Payments](/docs/api/payments) — 5 API operations tagged `payments`.\n- [Referrals](/docs/api/referrals) — 4 API operations tagged `referrals`.\n- [Reports](/docs/api/reports) — 1 API operation tagged `reports`.\n- [Resource Feedback](/docs/api/resource-feedback) — 5 API operations tagged `resource_feedback`.\n- [Resources](/docs/api/resources) — 6 API operations tagged `resources`.\n- [Runners](/docs/api/runners) — 6 API operations tagged `runners`.\n- [Setup](/docs/api/setup) — 5 API operations tagged `setup`.\n\n## CLI Reference\n\n24 command groups, generated from the argparse tree.\n\n- [Overview](/docs/cli/overview) — 24 command groups, generated from the argparse tree.\n- [logion admin](/docs/cli/admin) — 13 CLI commands.\n- [logion bounties](/docs/cli/bounties) — 19 CLI commands.\n- [logion completion](/docs/cli/completion) — 1 CLI command.\n- [logion course-reviews](/docs/cli/course-reviews) — 5 CLI commands.\n- [logion courses](/docs/cli/courses) — 26 CLI commands.\n- [logion credits](/docs/cli/credits) — 5 CLI commands.\n- [logion docs](/docs/cli/docs) — 1 CLI command.\n- [logion doctor](/docs/cli/doctor) — 1 CLI command.\n- [logion feedback](/docs/cli/feedback) — 3 CLI commands.\n- [logion health](/docs/cli/health) — 1 CLI command.\n- [logion identity](/docs/cli/identity) — 7 CLI commands.\n- [logion indexed](/docs/cli/indexed) — 1 CLI command.\n- [logion integrations](/docs/cli/integrations) — 4 CLI commands.\n- [logion listings](/docs/cli/listings) — 1 CLI command.\n- [logion notifications](/docs/cli/notifications) — 3 CLI commands.\n- [logion onboarding](/docs/cli/onboarding) — 1 CLI command.\n- [logion payments](/docs/cli/payments) — 5 CLI commands.\n- [logion recall](/docs/cli/recall) — 2 CLI commands.\n- [logion referrals](/docs/cli/referrals) — 4 CLI commands.\n- [logion reports](/docs/cli/reports) — 1 CLI command.\n- [logion resources](/docs/cli/resources) — 8 CLI commands.\n- [logion skills](/docs/cli/skills) — 11 CLI commands.\n- [logion update](/docs/cli/update) — 1 CLI command.\n- [logion usage](/docs/cli/usage) — 4 CLI commands.\n\n## About\n\nHow this documentation stays true.\n\n- [How these docs are built](/docs/reference/how-these-docs-are-built) — The three sources, the cross-link chain, and why staleness is a build failure.\n", + "body": "# Logion documentation\n\nLogion measures whether the skills, plugins, MCP servers and models an agent installs actually do what they claim, and publishes the method so anyone can reproduce it.\n\nTwo references and a set of guides. The references are generated from the contract and the CLI itself, so they cannot drift — see [How these docs are built](/docs/reference/how-these-docs-are-built).\n\n## Guides\n\nWritten by hand. Start here.\n\n- [Bounties and Referrals](/docs/guides/bounties-and-referrals) — Understand credit-funded bounties and referral rewards.\n- [Core Concepts](/docs/guides/concepts) — Understand courses, versions, capabilities, entitlements, and roles.\n- [Creating Courses](/docs/guides/creating-courses) — Author, declare, upload, review, and publish a course safely.\n- [Credits and Purchases](/docs/guides/credits-and-purchases) — Learn how credits, free courses, paid purchases, and approvals work.\n- [Getting Started](/docs/guides/getting-started) — Install, authenticate, discover, acquire, and use your first course.\n- [Marketplace Loop](/docs/guides/marketplace-loop) — The complete Logion marketplace loop: search, inspect, acquire, install, review, and bounty.\n- [Course Reviews](/docs/guides/reviews) — File honest agent reviews automatically after meaningful course use.\n- [Safety and Trust](/docs/guides/safety) — Inspect permissions, understand review limits, and report unsafe content.\n- [Use Observation and Feedback](/docs/guides/use-observation-and-feedback) — Observe use of natively installed resources and report honest feedback under an explicit consent mode.\n\n## API Reference\n\n129 operations across 28 groups, generated from the OpenAPI contract.\n\n- [Overview](/docs/api/overview) — 129 operations across 28 groups, generated from the OpenAPI contract.\n- [Admin · Agents](/docs/api/admin-agents) — 3 API operations tagged `admin-agents`.\n- [Admin · Bounties](/docs/api/admin-bounties) — 4 API operations tagged `admin-bounties`.\n- [Admin · Courses](/docs/api/admin-courses) — 3 API operations tagged `admin-courses`.\n- [Admin · Indexing](/docs/api/admin-indexing) — 8 API operations tagged `admin-indexing`.\n- [Admin · Referrals](/docs/api/admin-referrals) — 1 API operation tagged `admin-referrals`.\n- [Admin · Reports](/docs/api/admin-reports) — 4 API operations tagged `admin-reports`.\n- [Admin · Users](/docs/api/admin-users) — 3 API operations tagged `admin-users`.\n- [AI Catalog](/docs/api/ai-catalog) — 1 API operation tagged `ai-catalog`.\n- [ARD Discovery](/docs/api/ard) — 5 API operations tagged `ard`.\n- [Bounties](/docs/api/bounties) — 15 API operations tagged `bounties`.\n- [Capabilities](/docs/api/capabilities) — 1 API operation tagged `capabilities`.\n- [Course Reviews](/docs/api/course-reviews) — 7 API operations tagged `course-reviews`.\n- [Courses](/docs/api/courses) — 15 API operations tagged `courses`.\n- [Credits](/docs/api/credits) — 4 API operations tagged `credits`.\n- [Evals](/docs/api/evals) — 4 API operations tagged `evals`.\n- [Executions](/docs/api/executions) — 4 API operations tagged `executions`.\n- [General](/docs/api/general) — 2 API operations tagged `general`.\n- [Identity](/docs/api/identity) — 9 API operations tagged `identity`.\n- [Indexed Listings](/docs/api/indexed-listings) — 1 API operation tagged `indexed-listings`.\n- [Listings](/docs/api/listings) — 1 API operation tagged `listings`.\n- [Notifications](/docs/api/notifications) — 2 API operations tagged `notifications`.\n- [Payments](/docs/api/payments) — 5 API operations tagged `payments`.\n- [Referrals](/docs/api/referrals) — 4 API operations tagged `referrals`.\n- [Reports](/docs/api/reports) — 1 API operation tagged `reports`.\n- [Resource Feedback](/docs/api/resource-feedback) — 5 API operations tagged `resource_feedback`.\n- [Resources](/docs/api/resources) — 6 API operations tagged `resources`.\n- [Runners](/docs/api/runners) — 6 API operations tagged `runners`.\n- [Setup](/docs/api/setup) — 5 API operations tagged `setup`.\n\n## CLI Reference\n\n25 command groups, generated from the argparse tree.\n\n- [Overview](/docs/cli/overview) — 25 command groups, generated from the argparse tree.\n- [logion admin](/docs/cli/admin) — 13 CLI commands.\n- [logion bounties](/docs/cli/bounties) — 19 CLI commands.\n- [logion completion](/docs/cli/completion) — 1 CLI command.\n- [logion course-reviews](/docs/cli/course-reviews) — 5 CLI commands.\n- [logion courses](/docs/cli/courses) — 26 CLI commands.\n- [logion credits](/docs/cli/credits) — 5 CLI commands.\n- [logion docs](/docs/cli/docs) — 1 CLI command.\n- [logion doctor](/docs/cli/doctor) — 1 CLI command.\n- [logion eval](/docs/cli/eval) — 6 CLI commands.\n- [logion feedback](/docs/cli/feedback) — 3 CLI commands.\n- [logion health](/docs/cli/health) — 1 CLI command.\n- [logion identity](/docs/cli/identity) — 7 CLI commands.\n- [logion indexed](/docs/cli/indexed) — 1 CLI command.\n- [logion integrations](/docs/cli/integrations) — 4 CLI commands.\n- [logion listings](/docs/cli/listings) — 1 CLI command.\n- [logion notifications](/docs/cli/notifications) — 3 CLI commands.\n- [logion onboarding](/docs/cli/onboarding) — 1 CLI command.\n- [logion payments](/docs/cli/payments) — 5 CLI commands.\n- [logion recall](/docs/cli/recall) — 2 CLI commands.\n- [logion referrals](/docs/cli/referrals) — 4 CLI commands.\n- [logion reports](/docs/cli/reports) — 1 CLI command.\n- [logion resources](/docs/cli/resources) — 8 CLI commands.\n- [logion skills](/docs/cli/skills) — 11 CLI commands.\n- [logion update](/docs/cli/update) — 1 CLI command.\n- [logion usage](/docs/cli/usage) — 4 CLI commands.\n\n## About\n\nHow this documentation stays true.\n\n- [How these docs are built](/docs/reference/how-these-docs-are-built) — The three sources, the cross-link chain, and why staleness is a build failure.\n", "kind": "index", "slug": "index", "summary": "Guides, API reference, and CLI reference.", @@ -667,7 +674,7 @@ "pages": [ { "slug": "cli/overview", - "summary": "24 command groups, generated from the argparse tree.", + "summary": "25 command groups, generated from the argparse tree.", "title": "Overview" }, { @@ -710,6 +717,11 @@ "summary": "1 CLI command.", "title": "logion doctor" }, + { + "slug": "cli/eval", + "summary": "6 CLI commands.", + "title": "logion eval" + }, { "slug": "cli/feedback", "summary": "3 CLI commands.", @@ -791,7 +803,7 @@ "title": "logion usage" } ], - "summary": "24 command groups, generated from the argparse tree.", + "summary": "25 command groups, generated from the argparse tree.", "title": "CLI Reference" }, { @@ -809,7 +821,7 @@ ], "source": { "api_major": 1, - "cli_commands": 128, + "cli_commands": 134, "cli_version": "0.2.0.dev1", "contract_digest": "sha256:996d7d6160fd0c3cb9fdb4f4a2e4422319167c9617c958c333056c8416fc1d94", "linked_operations": 66, diff --git a/packages/runner/logion_runner/evals/__init__.py b/packages/runner/logion_runner/evals/__init__.py index 4c6930a6..011ea585 100644 --- a/packages/runner/logion_runner/evals/__init__.py +++ b/packages/runner/logion_runner/evals/__init__.py @@ -199,3 +199,25 @@ def _contract_digest(contract: EvalContract) -> str: from logion_eval_contract import contract_digest as _digest return _digest(contract) + + +# Imported after the adapter definitions because the executor consumes this +# public adapter surface while it initializes. +from logion_runner.evals.executor import ( # noqa: E402 + EvalExecutionError, + GradedOutcome, + execute_eval_contract, +) + +__all__ = [ + "EVAL_JOB_TYPE", + "EvalAdapterError", + "EvalExecutionError", + "GradedOutcome", + "ResolvedEvalJob", + "environment_for", + "execute_eval_contract", + "normalize_outcome", + "resolve_eval_job", + "subject_digest_for", +] diff --git a/packages/runner/logion_runner/evals/executor.py b/packages/runner/logion_runner/evals/executor.py index 7d8359a1..b2bfeb84 100644 --- a/packages/runner/logion_runner/evals/executor.py +++ b/packages/runner/logion_runner/evals/executor.py @@ -15,12 +15,14 @@ import time from dataclasses import dataclass from pathlib import Path -from typing import Protocol +from typing import Protocol, cast from logion_eval_contract import ( EvalContract, JsonObject, + JsonValue, ResultEnvironment, + canonicalize_text, ) from logion_eval_contract.models import AssertionOutcome, MetricValue @@ -115,6 +117,8 @@ def _assertion_holds( return _string_holds(operator, expected, observed) if ( operator in ("lt", "lte", "gt", "gte") + and not isinstance(observed, bool) + and not isinstance(expected, bool) and isinstance(observed, (int, float)) and isinstance(expected, (int, float)) ): @@ -127,7 +131,7 @@ def _assertion_holds( def _scalarize(observed: object) -> str | int | float | bool | None: if isinstance(observed, (str, int, float, bool)) or observed is None: return observed - return json.dumps(observed, sort_keys=True) + return canonicalize_text(cast(JsonValue, observed)) def _metric_observation( @@ -306,9 +310,14 @@ def _grade( observed = metrics_by_id[assertion.metric].value else: observed = _scalarize(observed_document.get(assertion.metric)) - holds = _assertion_holds( - assertion.operator, assertion.expected, observed - ) + try: + holds = _assertion_holds( + assertion.operator, assertion.expected, observed + ) + except (ValueError, re.error) as exc: + raise EvalExecutionError( + f"cannot evaluate assertion {assertion.id!r}: {exc}" + ) from exc outcomes.append( AssertionOutcome( id=assertion.id, diff --git a/scripts/sf.py b/scripts/sf.py new file mode 100644 index 00000000..5050e224 --- /dev/null +++ b/scripts/sf.py @@ -0,0 +1,99 @@ +#!/usr/bin/env python3 +"""Run the reviewed software-factory revision, never an sf found on PATH.""" + +from __future__ import annotations + +import hashlib +import json +import os +import subprocess +import sys +import tomllib +from pathlib import Path + +REVISION = "b06be44f6c982dac58b898778d7dba224d9ed7b1" +SOURCE = "https://github.com/nicolasmelo1/software-factory" +ROOT = Path(__file__).resolve().parents[1] +INSTALL = ROOT / ".local" / "software-factory" / REVISION + + +def digest(binary: Path) -> str: + return hashlib.sha256(binary.read_bytes()).hexdigest() + + +def verify_source(install: Path) -> None: + metadata = tomllib.loads((install / ".crates.toml").read_text()) + expected = f"(git+{SOURCE}?rev={REVISION}#{REVISION})" + owners = [key for key, bins in metadata["v1"].items() if "sf" in bins] + if len(owners) != 1 or not owners[0].endswith(expected): + raise ValueError( + "sf Cargo provenance does not match the reviewed revision" + ) + + +def verify(install: Path) -> Path: + binary = install / "bin" / "sf" + verify_source(install) + receipt = json.loads((install / "verified.json").read_text()) + if receipt != {"revision": REVISION, "sha256": digest(binary)}: + raise ValueError("sf binary differs from its verified installation") + if binary.is_symlink() or not os.access(binary, os.X_OK): + raise ValueError( + "sf must be an executable regular installation, not a symlink" + ) + return binary + + +def ensure_installed(install: Path) -> Path: + if not install.exists(): + # Partial installations stay untrusted; never fall back to PATH. + install.mkdir(parents=True) + subprocess.run( + [ + "cargo", + "install", + "--git", + SOURCE, + "--rev", + REVISION, + "--locked", + "--root", + str(install), + "--bin", + "sf", + ], + check=True, + ) + verify_source(install) + (install / "verified.json").write_text( + json.dumps({ + "revision": REVISION, + "sha256": digest(install / "bin" / "sf"), + }) + ) + return verify(install) + + +def main() -> int: + try: + binary = ensure_installed(INSTALL) + return subprocess.run( + [str(binary), *sys.argv[1:]], cwd=ROOT + ).returncode + except ( + OSError, + ValueError, + KeyError, + TypeError, + subprocess.CalledProcessError, + ) as exc: + print(f"Pinned sf unavailable: {exc}", file=sys.stderr) + print( + f"Remove {INSTALL} and retry to rebuild; PATH sf is never used.", + file=sys.stderr, + ) + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_sf_launcher.py b/tests/test_sf_launcher.py new file mode 100644 index 00000000..e4f471c0 --- /dev/null +++ b/tests/test_sf_launcher.py @@ -0,0 +1,143 @@ +"""The factory gate must not inherit an unrelated sf from the shell.""" + +from __future__ import annotations + +import importlib.util +import json +import subprocess +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[1] +SPEC = importlib.util.spec_from_file_location( + "pinned_sf", ROOT / "scripts" / "sf.py" +) +assert SPEC is not None +assert SPEC.loader is not None +sf = importlib.util.module_from_spec(SPEC) +SPEC.loader.exec_module(sf) + + +def installation(path: Path, revision: str = sf.REVISION) -> Path: + binary = path / "bin" / "sf" + binary.parent.mkdir(parents=True) + binary.write_text("#!/bin/sh\nexit 23\n") + binary.chmod(0o755) + (path / ".crates.toml").write_text( + f'[v1]\n"sf 0.2.0 ' + f'(git+{sf.SOURCE}?rev={revision}#{revision})" = ["sf"]\n' + ) + (path / "verified.json").write_text( + json.dumps({"revision": revision, "sha256": sf.digest(binary)}) + ) + return binary + + +def test_path_sf_cannot_override_pinned_binary(tmp_path, monkeypatch): + impostor = tmp_path / "sf" + marker = tmp_path / "path-sf-ran" + impostor.write_text(f"#!/bin/sh\ntouch '{marker}'\nexit 0\n") + impostor.chmod(0o755) + monkeypatch.setenv("PATH", str(tmp_path)) + install = tmp_path / "pinned" + installation(install) + monkeypatch.setattr(sf, "INSTALL", install) + monkeypatch.setattr(sys, "argv", ["sf.py", "check", "--allow-commands"]) + assert sf.main() == 23 + assert not marker.exists() + + +@pytest.mark.parametrize( + "damage", ["revision", "binary", "receipt", "metadata", "symlink"] +) +def test_unverifiable_installation_fails_closed(tmp_path, monkeypatch, damage): + install = tmp_path / "pinned" + binary = installation( + install, "0" * 40 if damage == "revision" else sf.REVISION + ) + if damage == "binary": + binary.write_text("#!/bin/sh\nexit 0\n") + elif damage == "receipt": + (install / "verified.json").unlink() + elif damage == "metadata": + (install / ".crates.toml").write_text("not valid TOML") + elif damage == "symlink": + target = tmp_path / "replacement" + binary.rename(target) + binary.symlink_to(target) + monkeypatch.setattr(sf, "INSTALL", install) + + def forbidden_run(*_args, **_kwargs): + pytest.fail("unverified sf must never execute or silently reinstall") + + monkeypatch.setattr(subprocess, "run", forbidden_run) + assert sf.main() == 1 + + +def test_install_uses_exact_revision_and_isolated_root(tmp_path, monkeypatch): + install = tmp_path / "pinned" + calls = [] + + def cargo_run(command, **kwargs): + calls.append(command) + assert kwargs == {"check": True} + installation(install) + (install / "verified.json").unlink() + + monkeypatch.setattr(subprocess, "run", cargo_run) + assert sf.ensure_installed(install) == install / "bin" / "sf" + assert calls == [ + [ + "cargo", + "install", + "--git", + sf.SOURCE, + "--rev", + sf.REVISION, + "--locked", + "--root", + str(install), + "--bin", + "sf", + ] + ] + sf.ensure_installed(install) + assert len(calls) == 1 + + +def test_failed_install_never_falls_back_to_path(tmp_path, monkeypatch): + monkeypatch.setattr(sf, "INSTALL", tmp_path / "pinned") + calls = [] + + def failing_cargo(command, **_kwargs): + calls.append(command) + raise subprocess.CalledProcessError(1, command) + + monkeypatch.setattr(subprocess, "run", failing_cargo) + assert sf.main() == 1 + assert len(calls) == 1 + assert calls[0][0] == "cargo" + + +def test_ci_and_make_share_launcher_and_command_permission(): + makefile = (ROOT / "Makefile").read_text() + assert "python3 scripts/sf.py verify --allow-commands" in makefile + assert "python3 scripts/sf.py check --allow-commands" in makefile + workflow = (ROOT / ".github/workflows/pr-safety.yml").read_text() + assert "run: make factory-check" in workflow + assert "run: sf " not in workflow + assert "cargo install" not in workflow + + +def test_actor_goal_requirement_keeps_its_documented_reason(): + docs = (ROOT / "docs/factory-rules.md").read_text() + section = docs.split("### L3.EVERY_ACTOR_HAS_A_GOAL\n", 1)[1].split( + "\n## ", 1 + )[0] + assert "each agent that declares a `driver`" in section + assert "whose `goal` is non-empty" in section + assert "**Why.**" in section + assert "**Fix.**" in section + assert "scripts/allowed_mute_actors.txt" in section diff --git a/uv.lock b/uv.lock index 5bcb9425..8547fc0b 100644 --- a/uv.lock +++ b/uv.lock @@ -1721,6 +1721,7 @@ source = { editable = "packages/cli" } dependencies = [ { name = "cryptography" }, { name = "logion-client" }, + { name = "logion-runner" }, { name = "logion-skillmap" }, { name = "pydantic" }, { name = "pyyaml" }, @@ -1744,6 +1745,7 @@ dev = [ requires-dist = [ { name = "cryptography", specifier = ">=50.0.0,<51.0.0" }, { name = "logion-client", editable = "packages/client" }, + { name = "logion-runner", editable = "packages/runner" }, { name = "logion-skillmap", editable = "packages/skillmap" }, { name = "pydantic", specifier = ">=2.7,<3.0.0" }, { name = "pyyaml", specifier = ">=6.0,<7.0" },