From 3b955e8fb834254fdd9922e5606df66a3b2adbcc Mon Sep 17 00:00:00 2001 From: logion-roadmap-bot Date: Wed, 9 Sep 2026 02:46:53 +0000 Subject: [PATCH] docs(roadmap): sync canonical planning Signed-off-by: logion-roadmap-bot --- docs/roadmap-sync-manifest.json | 78 +- .../human-dashboard-and-user-policy.md | 4 +- ...en-protocol-and-entitlement-portability.md | 8 +- plans/consumption-adoption-ladder.md | 6 +- ...driven-actor-performs-the-measured-flow.md | 174 +++++ plans/next-steps.md | 213 ++++- plans/normative-carry-overs.md | 170 +++- ...isition-artifact-delivery-and-inventory.md | 475 ------------ ...k-harness-adapter-and-logion-dsh-plugin.md | 396 ---------- ...observation-linked-feedback-and-reviews.md | 648 ---------------- ...lisher-integrated-consented-observation.md | 734 ------------------ ...i-catalog-publication-and-ard-discovery.md | 428 ---------- ...local-multi-agent-first-node-foundation.md | 33 +- .../phase-15.15-isolated-first-runner-node.md | 27 +- ...16.1-eval-contract-and-reference-runner.md | 406 +++++++++- .../phase-16.1.1-a-gate-outlives-its-plan.md | 213 +++++ ...phase-16.1.2-a-plan-is-one-pull-request.md | 208 +++++ ...e-documentation-is-read-out-of-the-code.md | 183 +++++ ...phase-16.1.4-a-published-entry-resolves.md | 201 +++++ ....5-what-is-installed-here-has-a-version.md | 231 ++++++ ...valuators-and-skill-reference-evaluator.md | 9 +- ...deterministic-replication-and-agreement.md | 3 +- ...l-attestations-and-cross-node-authority.md | 3 +- ...ase-16.9-benchmark-field-reconciliation.md | 4 +- plans/positioning-and-independence.md | 231 +++++- plans/public-answer-surface.md | 237 ++++++ plans/release-0.2.md | 304 +++++++- 27 files changed, 2777 insertions(+), 2850 deletions(-) create mode 100644 plans/driven-actor-performs-the-measured-flow.md delete mode 100644 plans/phase-15.10-native-acquisition-artifact-delivery-and-inventory.md delete mode 100644 plans/phase-15.10.1-deepseek-harness-adapter-and-logion-dsh-plugin.md delete mode 100644 plans/phase-15.11-native-use-observation-linked-feedback-and-reviews.md delete mode 100644 plans/phase-15.11.1-publisher-integrated-consented-observation.md delete mode 100644 plans/phase-15.12-ai-catalog-publication-and-ard-discovery.md create mode 100644 plans/phase-16.1.1-a-gate-outlives-its-plan.md create mode 100644 plans/phase-16.1.2-a-plan-is-one-pull-request.md create mode 100644 plans/phase-16.1.3-the-documentation-is-read-out-of-the-code.md create mode 100644 plans/phase-16.1.4-a-published-entry-resolves.md create mode 100644 plans/phase-16.1.5-what-is-installed-here-has-a-version.md create mode 100644 plans/public-answer-surface.md diff --git a/docs/roadmap-sync-manifest.json b/docs/roadmap-sync-manifest.json index 5939ef6a..32a1ee13 100644 --- a/docs/roadmap-sync-manifest.json +++ b/docs/roadmap-sync-manifest.json @@ -30,7 +30,7 @@ }, { "path": "future-roadmap/human-dashboard-and-user-policy.md", - "sha256": "3c456f7b-bf85e2ff-87f1713d-d3ea1c69-4630bd55-64ded856-2381ef14-74025dbe" + "sha256": "cfcebc9b-7438c096-6c1e74c0-f77a5c9b-c1ec28ce-80f83943-7ac2989c-1ccb6fb8" }, { "path": "future-roadmap/native-resource-feedback-and-first-node.md", @@ -38,7 +38,7 @@ }, { "path": "future-roadmap/open-protocol-and-entitlement-portability.md", - "sha256": "de8e1480-5668be79-9a40c0e7-4e0dd8c4-54c81036-7cfcfc1c-8820fb0f-05fe2218" + "sha256": "74c8c37a-79de460d-a0673ad5-90c31faa-7acac1c4-4e880537-c8958e16-18356354" }, { "path": "future-roadmap/post-launch-strategy.md", @@ -70,12 +70,16 @@ }, { "path": "plans/consumption-adoption-ladder.md", - "sha256": "b814ba20-d9e27f51-36398919-14206993-a1efa607-ba190225-8045057e-3bc1abe5" + "sha256": "2d1c7353-86b6277a-e34b9940-9c62dcd4-f0b6332f-8cd99105-98e0d111-ef93adc9" }, { "path": "plans/cycle-0-contract-e2e-hardening.md", "sha256": "28506ec7-5aaaddd0-76eb5a9b-0608e2a9-cf6577a7-662441bd-757d02ff-c9331c62" }, + { + "path": "plans/driven-actor-performs-the-measured-flow.md", + "sha256": "170ce4b2-279d76c7-8ea53021-544b4694-d337c150-edef5cb4-bca4fed5-3d81d841" + }, { "path": "plans/indexer-run-progress-observability.md", "sha256": "33c1fc21-7d214fa6-375ecbcc-28a474a1-9e0395f3-180589b0-f90e3fb3-acae8140" @@ -86,36 +90,16 @@ }, { "path": "plans/next-steps.md", - "sha256": "31db3ea7-09fe50cd-8bd71a21-6b47870d-d2ae5cc6-e7bc2f7c-576de2bb-54c52a33" + "sha256": "e20d7332-a39c171d-3b68bc69-0c6ae6ef-9e46cc0a-1b801195-24dde668-b356d635" }, { "path": "plans/normative-carry-overs.md", - "sha256": "18b61ccc-7d166dd2-b44840d0-47e609f9-3835ee6b-bf6aedf8-033c303c-64171dca" + "sha256": "e8cd98f1-ab388ea6-95afa76e-15df1274-37888b8f-df7c1187-5dc36934-a7bafb7e" }, { "path": "plans/phase-15-native-resource-loop-and-first-ai-catalog-ard-node.md", "sha256": "d8ec7a9a-791b23b5-8570fb90-e0f30cae-a1fbbd79-5e0a405e-2381688f-e17277c2" }, - { - "path": "plans/phase-15.10-native-acquisition-artifact-delivery-and-inventory.md", - "sha256": "79fcaf3e-09cbcad4-161eb37e-b14bd71b-b7878bba-aceb7bdb-0f30ca09-9c3af138" - }, - { - "path": "plans/phase-15.10.1-deepseek-harness-adapter-and-logion-dsh-plugin.md", - "sha256": "dcfb04e8-758b829f-37c292ff-dbcd18a2-bb31ca12-7d5b0f1c-02d5d57c-85d103b7" - }, - { - "path": "plans/phase-15.11-native-use-observation-linked-feedback-and-reviews.md", - "sha256": "1c0bc904-75836003-6811c3e4-c8bcfc4e-abe8730e-9e969e3b-84988eda-9acd0139" - }, - { - "path": "plans/phase-15.11.1-publisher-integrated-consented-observation.md", - "sha256": "31926669-fd250129-f9628228-560ffff0-ab0d7b0e-3776d0c0-c99cab59-64a0ce1a" - }, - { - "path": "plans/phase-15.12-ai-catalog-publication-and-ard-discovery.md", - "sha256": "cc6c5481-2a0af01e-0b363d03-e9a6940b-3349e324-a0e6996a-70f85ad6-84acb564" - }, { "path": "plans/phase-15.13-portable-scan-evidence.md", "sha256": "be9ed2a3-86a98f74-a9df1d7f-43e7d551-4619ccb4-460bcda9-3e511e4b-049cd266" @@ -126,11 +110,11 @@ }, { "path": "plans/phase-15.14.1-local-multi-agent-first-node-foundation.md", - "sha256": "a396ddad-4164d54c-b39dab94-31da0ae9-e9e68941-da2f8f84-e00c6cee-b3c4978d" + "sha256": "a0969006-97ae11fb-7d4151fe-065c0157-e3b6108f-5b0406ba-971e3881-86455af8" }, { "path": "plans/phase-15.15-isolated-first-runner-node.md", - "sha256": "b27bca03-16f928e3-ea74d387-70026145-4f70b119-3c83f80b-cf177403-41882042" + "sha256": "ad7fe673-e59032db-344a3bb5-2c255cba-095ac234-ec9152fe-ecfde678-e131eef8" }, { "path": "plans/phase-15.16-first-party-resource-dogfood-loop.md", @@ -146,7 +130,27 @@ }, { "path": "plans/phase-16.1-eval-contract-and-reference-runner.md", - "sha256": "9cb69e3c-de28f259-ddff0bac-ddbc10c6-725a28e3-ac77cb3c-ae1ed4bc-3d2767cb" + "sha256": "b449c423-95d36ef1-bed41c31-9e2f75be-121660b5-2938d8e1-223e1a77-6a116b2e" + }, + { + "path": "plans/phase-16.1.1-a-gate-outlives-its-plan.md", + "sha256": "481cbf84-d4e2d180-dce1f1f7-031430ac-b65c7dfc-9e93a69f-a2cf40f7-16415f7c" + }, + { + "path": "plans/phase-16.1.2-a-plan-is-one-pull-request.md", + "sha256": "68e01cfc-4f564ac4-37f533da-2ba32b40-0dcb458e-796159fb-ecc4327d-443e6f22" + }, + { + "path": "plans/phase-16.1.3-the-documentation-is-read-out-of-the-code.md", + "sha256": "277772c9-88bc61b3-76fa2489-897e38fb-9a91e238-f64b7b40-c631e456-6109075d" + }, + { + "path": "plans/phase-16.1.4-a-published-entry-resolves.md", + "sha256": "6fd740f4-eff58d30-19c61210-6ebe1326-3208798f-3e4b15a6-813d4919-81916c52" + }, + { + "path": "plans/phase-16.1.5-what-is-installed-here-has-a-version.md", + "sha256": "4c8a557a-e4906b01-203fc3c7-8680d7e5-810ec96b-c806f8e3-cf26be95-64abe66a" }, { "path": "plans/phase-16.10-external-runner-onboarding-and-conformance.md", @@ -162,7 +166,7 @@ }, { "path": "plans/phase-16.2-typed-evaluators-and-skill-reference-evaluator.md", - "sha256": "8674489b-651863c5-4cbd4cd3-2f2ce452-dca6a2ad-b4e2f5e0-ced0e3ec-d9eb0fb0" + "sha256": "bf2de90b-a8b7cc0b-6933d6c8-a73dcdcd-c41344a7-b3533208-1ddd7260-fbc0ec12" }, { "path": "plans/phase-16.3-runner-registry-and-network-jobs.md", @@ -170,11 +174,11 @@ }, { "path": "plans/phase-16.4-deterministic-replication-and-agreement.md", - "sha256": "bb136f7d-f0f6fcfb-515bf968-fa232b36-e8aa3506-51bc4260-fc632582-45ffaa31" + "sha256": "cc1ac948-8b9bd2bb-f2b358f5-bcd2e239-1661e506-f10d0ed7-62a36cdd-0aba6b4a" }, { "path": "plans/phase-16.5-eval-attestations-and-cross-node-authority.md", - "sha256": "efec965f-00f2095d-f75c4f14-98063494-e76eee2f-148f2b66-d91eb657-54d84820" + "sha256": "e16f0917-64824862-da76fe17-5bb92fc7-09e26e8b-5571782d-e21a073a-dc6801cc" }, { "path": "plans/phase-16.6-benchmark-backed-bounties.md", @@ -190,7 +194,7 @@ }, { "path": "plans/phase-16.9-benchmark-field-reconciliation.md", - "sha256": "c09c8d3c-14298878-11354114-5fe37d3b-a65bb7ed-638881bd-8b91d6ed-423c95ae" + "sha256": "a3d377b7-c852cddd-5595a416-83c41849-ac669438-016e3903-48c70b86-0ccea50e" }, { "path": "plans/phase-17-open-ecosystem-and-production-hardening.md", @@ -226,11 +230,15 @@ }, { "path": "plans/positioning-and-independence.md", - "sha256": "0df78710-3faa4046-f5f8a600-97bda4c7-371a4a08-e95cf9df-3f51de51-4ca9b3bc" + "sha256": "58d813ac-81ec01ac-9ac9083d-9e18fc23-0743e84c-d4465fcf-7f78a338-91bd9be0" + }, + { + "path": "plans/public-answer-surface.md", + "sha256": "d86eaf49-8eedbe9c-0a6c1265-1c6ded22-d8c0f8e6-3912fe2a-9cfafb2c-44c3a1fd" }, { "path": "plans/release-0.2.md", - "sha256": "4a5f1f7b-06c7fe45-6f280905-91468f40-4b56b2aa-a261bdf6-efe008cd-5e9274db" + "sha256": "8b41801e-790d6cb2-fe397f99-cc7fc2d9-45485750-f2c64c60-97aa7840-5061dc9e" }, { "path": "protocol-specs/README.md", @@ -294,5 +302,5 @@ } ], "schema_version": 1, - "source_revision": "9f43c8f5-9534ee5b-644a30f6-a43c8dba-a987b9d2" + "source_revision": "31b8ea8b-1aab8817-b2534fd1-77a9e8c6-abd1c888" } diff --git a/future-roadmap/human-dashboard-and-user-policy.md b/future-roadmap/human-dashboard-and-user-policy.md index 60fe8a93..6750ff15 100644 --- a/future-roadmap/human-dashboard-and-user-policy.md +++ b/future-roadmap/human-dashboard-and-user-policy.md @@ -15,8 +15,8 @@ Neither ships in `logion` 0.2.0. The readiness and enforcement seed now lives in [`plans/phase-15.15-isolated-first-runner-node.md`](../plans/phase-15.15-isolated-first-runner-node.md) (runner doctor, project-scoped resolution, and sandbox enforcement). Native inventory plus `off|local-only|prompt|auto` feedback consent lands earlier -in [`plans/phase-15.10`](../plans/phase-15.10-native-acquisition-artifact-delivery-and-inventory.md) -and [`plans/phase-15.11`](../plans/phase-15.11-native-use-observation-linked-feedback-and-reviews.md); +in [`maintainer documentation: native-acquisition-and-inventory.md`](../maintainer documentation: native-acquisition-and-inventory.md) +and [`maintainer documentation: native-use-observation-and-feedback.md`](../maintainer documentation: native-use-observation-and-feedback.md); the dashboard must configure those existing policies rather than inventing a second telemetry preference. diff --git a/future-roadmap/open-protocol-and-entitlement-portability.md b/future-roadmap/open-protocol-and-entitlement-portability.md index 5b47b346..937b8e8a 100644 --- a/future-roadmap/open-protocol-and-entitlement-portability.md +++ b/future-roadmap/open-protocol-and-entitlement-portability.md @@ -36,7 +36,7 @@ operating the reference index/node; `sh.logion.*` namespaces, the `logion` CLI, and the Logion node's payment rails are that node's own identifiers and keep their names. Resource publication uses [AI Catalog](https://ai-catalog.io/) and discovery uses ARD -([`plans/phase-15.12`](../plans/phase-15.12-ai-catalog-publication-and-ard-discovery.md)); +([`maintainer documentation: ai-catalog-and-ard-discovery.md`](../maintainer documentation: ai-catalog-and-ard-discovery.md)); an optional catalog entry relation/discovery extension advertises the AKTP evidence/improvement feed ([`plans/phase-15.17`](../plans/phase-15.17-aktp-evidence-and-improvement-feed-v0.md)). @@ -62,7 +62,7 @@ operating the index is the position. | ATProto | Logion equivalent | Status today | | --- | --- | --- | -| PDS (self-hosted user data server, permissionless) | Resource/evidence node | AI Catalog publication + ARD discovery in [`plans/phase-15.12`](../plans/phase-15.12-ai-catalog-publication-and-ard-discovery.md), AKTP events in [`plans/phase-15.17`](../plans/phase-15.17-aktp-evidence-and-improvement-feed-v0.md) | +| PDS (self-hosted user data server, permissionless) | Resource/evidence node | AI Catalog publication + ARD discovery in [`maintainer documentation: ai-catalog-and-ard-discovery.md`](../maintainer documentation: ai-catalog-and-ard-discovery.md), AKTP events in [`plans/phase-15.17`](../plans/phase-15.17-aktp-evidence-and-improvement-feed-v0.md) | | Relay (crawls all PDSs, aggregates) | The `logion-indexer` crawler feeding `api.logion.sh` | Shipped shape in phase 15.6 | | AppView (indexes, serves the app view, applies moderation) | `api.logion.sh` listings/search + the scanner pipeline as reach policy | Exists (centralized) | | Client (Graysky, deer.social, …) | `logion` CLI, agent-companion, third-party clients via the public wire contract | Resource CLI shipped as an additive surface in phase 15.9; white-label bins were dropped | @@ -141,7 +141,7 @@ evidence is what the coalition analysis in 1. **v0 — self-announce.** `POST /v1/nodes/announce {host}` on the index + `logion node announce` in the CLI (the requestCrawl pattern; spec and - abuse bounds were superseded by ARD ingestion in [`plans/phase-15.12`](../plans/phase-15.12-ai-catalog-publication-and-ard-discovery.md) + abuse bounds were superseded by ARD ingestion in [`maintainer documentation: ai-catalog-and-ard-discovery.md`](../maintainer documentation: ai-catalog-and-ard-discovery.md) §2.6). An announce buys one validated well-known fetch, never a listing. Seed yaml remains as bootstrap, never as gate. 2. **v1 — peer exchange.** Optional `known_nodes: []` (array of ≤64 @@ -270,7 +270,7 @@ buyer. Everything else is distribution. | Stage | What ships | Entitlement story | Plan/roadmap anchor | | --- | --- | --- | --- | -| v0 — indexed resources | AI Catalog publisher/consumer + ARD Agent Finder indexer + optional AKTP evidence/improvement link | **None needed.** Discovery is open; settlement remains node-local | [`plans/phase-15.12`](../plans/phase-15.12-ai-catalog-publication-and-ard-discovery.md), [`plans/phase-15.17`](../plans/phase-15.17-aktp-evidence-and-improvement-feed-v0.md) | +| v0 — indexed resources | AI Catalog publisher/consumer + ARD Agent Finder indexer + optional AKTP evidence/improvement link | **None needed.** Discovery is open; settlement remains node-local | [`maintainer documentation: ai-catalog-and-ard-discovery.md`](../maintainer documentation: ai-catalog-and-ard-discovery.md), [`plans/phase-15.17`](../plans/phase-15.17-aktp-evidence-and-improvement-feed-v0.md) | | v1 — signed feeds | Node keypairs, signed feed/manifests, `nodes` registry, `known_nodes` peer exchange, attestation read surface | Still link-out; signatures make mirroring/tamper-evidence real | Protocol-Ready 2A–2C | | v2 — portable entitlements | Entitlement credentials + revocation events + archived keys + CLI verification | The target model above; buy at a node with Pix, install anywhere | New spec work (this doc) | | v3 — federation | Node registry, sync, mirrors, trust levels, cross-node reads | Cross-node *payments* remain out of protocol even here | Protocol-Ready 2E | diff --git a/plans/consumption-adoption-ladder.md b/plans/consumption-adoption-ladder.md index 5dcd4bb3..521af930 100644 --- a/plans/consumption-adoption-ladder.md +++ b/plans/consumption-adoption-ladder.md @@ -29,7 +29,7 @@ hf download owner/model --revision COMMIT logion resources acquire RESOURCE_ID --version VERSION_ID --channel logion_bundle ``` -Logion may recommend or delegate to these commands, but it does not silently replace them. Phase [`15.10`](phase-15.10-native-acquisition-artifact-delivery-and-inventory.md) adds: +Logion may recommend or delegate to these commands, but it does not silently replace them. Shipped, per [`native-acquisition-and-inventory.md`](../maintainer documentation: native-acquisition-and-inventory.md): - real Logion-hosted Course/capability downloads; - native acquisition plans; @@ -50,12 +50,12 @@ npx plugins add OFFICIAL_LOGION_PLUGIN The skill installs the Logion companion into the same Agent Skills workflow the user already uses. The plugin installs the thin observer integration where supported. If the verified Logion CLI is absent, first use explains and requests approval for its official installer; neither native command silently installs a binary, enables hooks, uploads telemetry, or opts into automatic feedback. -Phase [`15.11`](phase-15.11-native-use-observation-linked-feedback-and-reviews.md) owns this surface. +[Native-use observation](../maintainer documentation: native-use-observation-and-feedback.md) owns this surface. ## Rung 1.1 — Accept Logion inside the publisher's resource A resource publisher may use -[`15.11.1`](phase-15.11.1-publisher-integrated-consented-observation.md) +[the publisher-integrated path](normative-carry-overs.md#publisher-integrated-observation--designed-not-built) to ship a thin, open-source observer in a native plugin projection. The user installs the resource through the manager they already use and sees one exact disclosure before any observation state or network request: diff --git a/plans/driven-actor-performs-the-measured-flow.md b/plans/driven-actor-performs-the-measured-flow.md new file mode 100644 index 00000000..7a34b4a0 --- /dev/null +++ b/plans/driven-actor-performs-the-measured-flow.md @@ -0,0 +1,174 @@ + + +# Driven actor performs the measured flow + +> **Honesty boundary:** a scenario's agent roster is what its gate claims met the +> product. Today, in two sealed scenarios, a ~1000-line Python script performs +> the flow and a driven agent stands next to it saying nothing. + +Status: open. Blocks nothing that is already merged; blocks the merge of +`logion#317` only if the driven-actor rule ships in that PR. + +Exit condition: `python3 scripts/check_scenario_actors.py` (in `logion`) exits 0 with +`isolated_runner_node.yaml:node_operator` and +`eval_contract_reference_runner.yaml:node_operator` absent from +`scripts/allowed_mute_actors.txt`, and the resealed +`artifacts/phase-gates/phase-15.15.json` and `phase-16.1.json` each record a +real-agent run whose `node_operator` phase carries a non-empty goal. + +## What is actually wrong + +The driven-actor rule — `scripts/check_scenario_actors.py`, new in +`logion#317` — reports two findings: + +``` +eval_contract_reference_runner.yaml:node_operator: declares a driver but no phase gives it a goal +isolated_runner_node.yaml:node_operator: declares a driver but no phase gives it a goal +``` + +Both scenarios have the same shape. `node_operator` declares `driver: codex`, +its phases carry `goal: ""`, and the work happens in a `local_hook`: + +| scenario | phase | hook | assertions | +| --- | --- | --- | --- | +| `isolated_runner_node` | `execute_runner_evidence` | `run_runner_evidence.py` (1066 lines) | 0 | +| | `collect_runner_evidence` | `capture_runner_evidence.py` | 7 | +| | `verify_api_logs` (actor `auditor`, real goal) | — | 1 | +| `eval_contract_reference_runner` | `execute_eval_evidence` | `run_eval_evidence.py` (857 lines) | 0 | +| | `collect_eval_evidence` | `capture_eval_evidence.py` | 8 | +| | `verify_api_logs` (actor `auditor`, real goal) | — | 1 | + +`isolated_runner_node` additionally lists `consumer`, `evaluator`, +`contributor` and `sponsor` as driven agents with no phase at all; those four +are already frozen in `scripts/allowed_mute_actors.txt`. So of six declared +driven roles, exactly one speaks, and its only job is `logs.no_500s`. + +The hooks drive the flow the assertions measure: `httpx` straight at the API for +enrolment, contract upload, result submission and the rejection classes, plus +`subprocess` for wheel builds, fresh venvs and Docker. That is the shape the +rule's `why` describes as "a replay filed as evidence about an outcome". + +## Two moves that look like fixes and are not + +**Do not add the two keys to `scripts/allowed_mute_actors.txt`.** That file's own +header forbids it: the baseline was taken from `main` at the commit that +introduced the rule, "deliberately not from the branch that added it: freezing +the branch would have grandfathered the two violations the rule was written to +catch." Adding a key there is the one move the file exists to make visible in +review. + +**Do not install the runner wheel into the devrig role tree.** Putting +`logion-node` on an agent's PATH via `scripts/devrig/build_artifacts.py` invents +a distribution path no customer has, which is exactly the customer-fidelity +violation [the phase gate](agent-proving-ground-phase-gate.md) forbids: "start +from the public entry point a customer would have". It is not needed either — +see the next section. + +## Verified facts, so this is not re-investigated + +Established 2026-09-04 against `logion` at `refs/remotes/pr/317`: + +- **`logion-node` is already on the role image's PATH, by a real path.** + `deploy/local-node/node.sh:159` runs `uv build --all-packages --wheel --out-dir + dist-wheels`; `packages/runner` is a workspace member declaring + `logion-node = "logion_runner.cli:main"`; `Dockerfile.role:19-20` copies + `dist-wheels/` and `pip install`s every wheel. This is why the compose + `runner` service can declare `entrypoint: ["logion-node"]` over the same + `LOGION_NODE_IMAGE` that `consumer` and `auditor` use. +- **`consumer` and `auditor` are long-running, so `docker compose exec` works.** + They inherit the `x-role-base` command (`timeout ... sleep infinity`). Only + the `runner` service is `run --once` and exits, and it sits behind + `profiles: [runner]`. The agent-operates-a-role idiom is already proven by + `local_multi_agent_node.yaml`'s `consumer_repo_scoped_install` phase. +- **No eval subcommand needs Docker.** `execute_eval_contract` uses + `LocalTestBackend(python_executable=sys.executable)` + (`packages/runner/logion_runner/evals/executor.py:246`), so + `logion-node eval validate | run | inspect-result | compare` all work inside a + read-only role container with a `tmpfs` `/tmp`. +- **The `logion` CLI has no eval surface at all.** Nothing under + `packages/cli/cli` matches `eval|runner|node`. Driving this through the + consumer CLI would mean designing and shipping a new command group, which + 16.1 did not ask for. +- **Four eval operations are in the public v1 contract**, so HTTP is a legitimate + customer entry point for the parts no CLI covers (the phase gate lists + "public HTTP API" among the acceptable entry points): + `POST /v1/evals/contracts` (`upload_eval_contract`), + `GET /v1/evals/contracts/{ref}` (`get_eval_contract`), + `POST /v1/evals/jobs/validate` (`validate_eval_job`), + `POST /v1/evals/results` (`submit_eval_result`). +- **A goal cannot name the golden contract.** It lives at + `packages/eval-contract/tests/fixtures/golden_contract.json`, and both + `packages/` and `tests/fixtures/` are in + `customer_fidelity.forbidden_goal_substrings`. The driven phase therefore + needs a seeding phase first, and the goal must name the in-container path the + seed created — the same construction `local_multi_agent_node.yaml` uses for + its fixture skill bundle. +- **`capture_eval_evidence.py` is only 103 lines** and merely assembles a + manifest from files the run script wrote. All the weight is in + `run_eval_evidence.py`. + +## Work breakdown + +The scenario edit and the hook rewrite are **one change, not two**: a real goal +requires seeded fixtures, and seeding is a change to the run script. + +1. **Split `run_eval_evidence.py` three ways.** A `seed` mode (build the + validator wheel, seed contract and subject into the operator's workspace) that + runs as a phase with no `driver`, so the scenario says plainly that setup is a + fixture. The middle — validate, upload, the two executions, compare, submit, + lookup, the five rejection classes — leaves the script. A `collect` mode reads + the outcome back and types the facts. +2. **Give `node_operator` one phase with a real goal.** One is enough: the + checker's `voiced` set is per agent, not per phase, so the existing + hook-only collect phases stay legal. Write it in the + `local_multi_agent_node.yaml` idiom — `docker compose --project-directory + deploy/local-node ... exec -T sh -c ''`, in-container paths + only, with a `success_hint`. +3. **Decide where each fact's authority lives.** This is the hard part and the + reason the change deserves its own mutation tests. Four of the eight 16.1 + assertions have server-side truth and must read it rather than trust the + agent. For facts that are only local (`validator_import_root`, + `validation_exit_code`), apply the provenance shape + `files.observation_from_live_hook` got in `ef710d4`: the transcript must name + the session the payload claims, must sit outside every root the agent can + write, and the payload must carry the harness's own event fields. Without + this step the change moves forgery from a script to an agent instead of + removing it. +4. **Repeat 1–3 for `run_runner_evidence.py`.** Its rig-only parts are genuinely + rig and must end up in undriven phases: the sandbox image build, host-side + canary planting, the expired-lease sweep, and one adversarial job per + forbidden effect. +5. **Reseal.** Editing either scenario YAML changes an activation path, so + `phase-15.15.json` and `phase-16.1.json` both go stale and each needs a fresh + real-agent run (`codex`/`gpt-5.4-mini`, `api_adapter: local-devrig`). + `phase-15.14.1.json` is unaffected — neither YAML is in its activation list. + +## Sequencing against the merge + +**Measured 2026-09-04, after `canonical maintainer workspace#172` and `backend repository#176` +merged and with `logion#317` still open:** + +| tree under audit | findings | +| --- | --- | +| `main` as it stands now | **4 CRITICAL** | +| the same, with `logion#317`'s tree | **0** | + +The four are the two stale seals (`15.14.1`, `15.15`) plus two the private merge +created: merging `#176` put `packages/api/api/evals/` on private `main`, which +activates `16.1` through its `private_any` path while the public leg has no +scenario and no seal (`PHASE_REQUIRED_SCENARIO_MISSING`, +`PHASE_REAL_EVIDENCE_MISSING`). + +So `main` is red until `logion#317` lands, and `logion#317` is the only thing +that makes it green. This plan cannot do it quickly. Land `#317` first with the +driven-actor rule removed, then do this work as its own change with the rule as +its exit criterion. Nothing the rule ships — +`.software-factory/rules/every-actor-has-a-goal.yaml`, +`scripts/check_scenario_actors.py`, `scripts/allowed_mute_actors.txt`, +`docs/factory-rules.md`, the mutation directory — is an activation path of +`15.14.1`, `15.15` or `16.1`, so removing it preserves all three fresh seals and +costs no real-agent run. + +Holding `#317` instead is no longer a neutral choice: two of the three legs are +already on `main`, so holding leaves four criticals on `main` for however long +this plan takes, not two. diff --git a/plans/next-steps.md b/plans/next-steps.md index 5fdb3024..887d39b6 100644 --- a/plans/next-steps.md +++ b/plans/next-steps.md @@ -9,19 +9,29 @@ Everything that is context, contract, or policy moved out on 2026-08-17. | --- | --- | | Cutting the next release; defining what "done" means | [`release-0.2.md`](release-0.2.md) | | Writing public copy, arguing positioning, touching protocols | [`positioning-and-independence.md`](positioning-and-independence.md) | +| Touching the landing, the report page, or any public answer | [`public-answer-surface.md`](public-answer-surface.md) | | Implementing identity, acquisition, reconciliation, observation | [`normative-carry-overs.md`](normative-carry-overs.md) | | Instructing counsel, forming the entity | [`legal-and-entity.md`](legal-and-entity.md) | | Publishing a measurement about someone else's artifact | [`../maintainer documentation: measurement-publication-playbook.md`](../maintainer documentation: measurement-publication-playbook.md) | -**Current position:** step 3 — phase 15.12 (AI Catalog publication and ARD -discovery) is being implemented on `feat/phase-15.12-ai-catalog-ard`. Nothing -has been published since 0.1.15 on 2026-07-19. +**Current position (2026-09-07):** step 5a — the plan cadence +(`16.1.1`–`16.1.3`). Step 5 sealed on 2026-09-07: `16.1` carries a passing gate +with no caveats, so a portable eval contract runs through the reference runner. +It cost seven pull requests and 14,443 inserted lines across three +repositories, which is what step 5a exists to stop repeating. Nothing has been +published since 0.1.15 on 2026-07-19. **The one-sentence goal:** *evaluation is the entry, observation is the upsell.* A controlled eval needs nobody's permission and works at N=1; consented field observation needs users and is what gets sold after an eval report opens the door. +**Corollary, recorded 2026-09-01:** the published measurement is the *acquisition* +surface and the CLI is *retention*. Local skill hygiene is not the wedge — the +pain is real but its carrier is an organisation, and an org is reached through +someone who already believes a number. See +[`positioning-and-independence.md` §The wedge](positioning-and-independence.md#the-wedge). + ## Execution order Authoritative. A phase not listed here is written, valid, and **off the critical @@ -34,15 +44,24 @@ their own exit conditions: [`asm-logion-collaboration-and-protocol-convergence-g work of its own), and [`positioning-and-independence.md`](positioning-and-independence.md) (binding language discipline, no work of its own). +**Steps 1–5d are before the publish. Steps 6–10 are after it.** Everything before +the line exists to make one sentence true — a person who is not Logion acquires +something, uses it, and that use becomes evidence a publisher can be shown. +Everything after the line is about that evidence being believed by someone else. + | # | Work | Exit condition | | --- | --- | --- | -| 1 | [`15.11`](phase-15.11-native-use-observation-linked-feedback-and-reviews.md) closeout | The remaining "Still open" items are closed and the observation hook fires **live**, not from a replayed payload. Item 1 and item 2 are also prerequisites of step 2 — see below. | +| 1 | `15.11` closeout | **Done.** Sealed gate `artifacts/phase-gates/phase-15.11.json`. Plan retired 2026-08-27; shipped shape in [`../maintainer documentation: native-use-observation-and-feedback.md`](../maintainer documentation: native-use-observation-and-feedback.md). Residual debt (unpublished companion/observer, cross-driver live hook) is in [`normative-carry-overs.md`](normative-carry-overs.md). | | 2 | **Cross-hub install reconciliation** | One canonical artifact + version shows where it is listed across every indexed hub, and the install counts of the hubs that publish them, with coverage and blind spots stated. No hub can answer this about another hub. | -| 3 | [`15.12`](phase-15.12-ai-catalog-publication-and-ard-discovery.md) | `logion.sh` serves `/.well-known/ai-catalog.json` and ARD ingestion runs. **ASM contact (T1) fires before this design freeze.** | -| 4 | [`15.14.1`](phase-15.14.1-local-multi-agent-first-node-foundation.md) + [`15.15`](phase-15.15-isolated-first-runner-node.md) | One isolated local node runs jobs in rootless containers. Hard prerequisite of 16.1. | -| 5 | [`16.1`](phase-16.1-eval-contract-and-reference-runner.md) + [`16.2`](phase-16.2-typed-evaluators-and-skill-reference-evaluator.md) | A portable eval contract runs a third-party skill through the reference runner and produces a normalized result. | +| 3 | `15.12` | **Done.** Sealed gate `artifacts/phase-gates/phase-15.12.json`, and `logion.sh` serves a valid `/.well-known/ai-catalog.json` as of 2026-08-27. Plan retired; shipped shape in [`../maintainer documentation: ai-catalog-and-ard-discovery.md`](../maintainer documentation: ai-catalog-and-ard-discovery.md). | +| 4 | [`15.14.1`](phase-15.14.1-local-multi-agent-first-node-foundation.md) + [`15.15`](phase-15.15-isolated-first-runner-node.md) | **Done 2026-09-01.** Sealed gates `artifacts/phase-gates/phase-15.14.1.json` and `phase-15.15.json`, both `passed` with no caveats. One isolated local node runs signed jobs in rootless containers; the hard prerequisite of 16.1 is met. Plans are **not** retired — they carry the live evidence contracts the 16.1 receipt path inherits. Residual debt: three 15.15 criteria have no check designed (see `DEFERRED.md`), 15.14.1's verdict is not recomputed from typed facts until the freeze lifts 2026-09-30, and the 15.15 dogfood protocol was not completed. | +| 5 | [`16.1`](phase-16.1-eval-contract-and-reference-runner.md) | A portable eval contract runs a third-party skill through the reference runner and produces a normalized result. Gate sealed 2026-09-07 (`artifacts/phase-gates/phase-16.1.json`, `passed`, no caveats). Two criteria still have no check designed (see `DEFERRED.md`) and step 5d reseals it from a run where the actor performs the flow. | +| 5a | [`16.1.1`](phase-16.1.1-a-gate-outlives-its-plan.md) + [`16.1.2`](phase-16.1.2-a-plan-is-one-pull-request.md) + [`16.1.3`](phase-16.1.3-the-documentation-is-read-out-of-the-code.md) | The plan cadence, before the next phase uses it. A delivered plan retires into `maintainer documentation: phases/` under its own number and `docs_integrity` holds it there; a plan promises something checkable or `sf check` is red; the indexes and command lists in the three READMEs are read out of the repositories instead of retyped. Sits between 16.1 and 16.2 on purpose — 16.1 is the phase that measured what a nine-assertion gate costs, and 16.2 is the first that can be built in numbered children instead. | +| 5b | [`16.2`](phase-16.2-typed-evaluators-and-skill-reference-evaluator.md) | Typed evaluators, entered as numbered children under the 16.1.2 ceiling rather than as one entry. | +| 5c | [`public-answer-surface.md`](public-answer-surface.md) | One URL answers "does this artifact work?" with no account and no install, renders at least one real finding on first load, and content-negotiates for an agent. Absorbs [`17.6`](phase-17.6-public-narrative-and-landing-truth-pass.md) and the report page — it is **less** 0.2 scope, not more. | +| 5d | [`driven-actor-performs-the-measured-flow.md`](driven-actor-performs-the-measured-flow.md) | The `node_operator` in `isolated_runner_node` and `eval_contract_reference_runner` carries a real goal and performs the flow its assertions measure, with `15.15` and `16.1` resealed from runs where that is true. Not a product step: it removes a reward-hacking shape from two already sealed gates. The driven-actor rule in `logion#317` is its exit criterion, which is why landing that rule before this work is what blocks a merge. | | — | **RELEASE 0.2** | Loops A, B, D work end to end. Full gate in [`release-0.2.md`](release-0.2.md). No publish before this. Note that Loop B's primary launch answer is Logion's own controlled evaluation, so measurements exist *before* this line, not after it. What step 6 adds is publishing one as a report about a named third party. | -| 6 | **First public measurement** | One published, reproducible evaluation of a third-party artifact: prose, stated method and limits, the subject version, and the command to reproduce. The real gate — not "16.2 merged", and not the same artifact as the in-product answer that 0.2 already ships. | +| 6 | **First public measurement** | One published, reproducible evaluation of a third-party artifact: prose, stated method and limits, the subject version, and the command to reproduce — **and one reproduction of it by somebody who is not Logion**. The real gate — not "16.2 merged", and not the same artifact as the in-product answer that 0.2 already ships. Chosen shape recorded [below](#the-shape-of-step-6); why the reproduction is part of the exit condition rather than a nice-to-have is in [`positioning-and-independence.md`](positioning-and-independence.md#reproduction-is-the-substitute-for-reputation). | | 7 | [`15.17`](phase-15.17-aktp-evidence-and-improvement-feed-v0.md) | AKTP v0, minimal, defined only after real payload exists, and the first `evidence.published` event carries the step 6 measurement. The event cannot be step 6's own exit condition, because the protocol that transports it is built here. | | 8 | Publisher outreach | Contact carries a finished report. Nobody is asked to install anything first. | | 9 | [`17.3`](phase-17.3-resource-claims-and-commercial-rails.md) + [`16.3`](phase-16.3-runner-registry-and-network-jobs.md)–[`16.5`](phase-16.5-eval-attestations-and-cross-node-authority.md) | Claims/commerce (Loop C) and federation. 16.3–16.5 are **blocked by definition** until an external operator exists. | @@ -99,6 +118,104 @@ who set `off` in one was not honoured by the other. Step 1 remains a prerequisite of nothing in step 2. Its own exit condition still stands on its own terms. +### Why the envelope fields land before the publish + +Two fields are still owed to the observation envelope: the **harness version**, +and a **model slug drawn from a closed allowlist**. The contract is in +[`normative-carry-overs.md`](normative-carry-overs.md#envelope-fields--decided-2026-08-27-unbuilt) +rather than in a phase file, because `15.11` is closed and its plan is being +retired — the decision has to outlive it. It attaches to whichever phase next +touches the envelope. + +The sequencing argument is about schema, not product. The spool rejects any +record whose `integration_version` does not match, so adding a field later is a +version bump that invalidates every installed hook. **The cheapest moment to +change a strictly-versioned envelope is while the installed base is approximately +zero**, which is now and will not be true after 0.2. Nothing else about these +fields is urgent; the schema discipline is what makes them urgent. + +Measured token counts are **not** added. The harness does not expose per-resource +token attribution in a hook payload, so any number there would be inferred, and +an inferred number is what this whole document forbids. Token measurement belongs +to `16.2`, where the runner controls the loop. The existing `token_efficiency` +score stays what it is: a declared opinion, not a measurement. + +### The shape of step 6 + +Step 6 is one published measurement. Its cheapest high-signal shape is an +**aggregate over a bounded top-N of one hub** — "we ran a published eval contract +over the N most-installed skills on ; X% do not do what they claim" — rather +than a single artifact in isolation. + +Three reasons this shape is preferred: + +1. It costs one bounded run, not a fleet, which is the binding constraint while + the founder funds compute personally. +2. It is the number this workspace already carries as unverified: the reported + *~67% of skills fail in practice* in + [`positioning-and-independence.md`](positioning-and-independence.md) is marked + "confirm before public use". Confirming it **is** the launch. +3. It satisfies both the publication rule and the outreach step at once: + **publish the aggregate, send each individual finding to its author + privately.** The aggregate is the public artifact; the individual reports are + step 8's outreach, which then arrives with a finished report as required. + +The observed pattern in this niche supports the shape — `skill-history.com` over +skills.sh, `apifystats.com` and `audit-tools.ai` over Apify. What travels is a +third party measuring the popular thing and publishing an uncomfortable number. +Yukon launched on one verifiable result, not an architecture. + +**One external reproduction is part of step 6, not a follow-up.** An unknown +issuer has no reputation to lend the number, and the answer to that is not to +accumulate reputation first — it is that a stranger can re-run the result in one +command. That property is only real once somebody has actually done it, so the +step does not close on publication. What counts: a person who is not Logion runs +the published command against the pinned subject version and reports the +result — agreeing *or* disagreeing. A disagreement that surfaces a real method +error is a better outcome than silence, and it is handled by Step 3 of +[the publication playbook](../maintainer documentation: measurement-publication-playbook.md). +Cheap, and it is the only down-payment on +[issuer #2](positioning-and-independence.md#issuer-2--the-milestone-that-makes-the-thesis-true) +available before federation exists. + +**Preferred first subject: a token-reduction skill.** This class carries a +*quantitative, falsifiable, publisher-declared* claim, which makes it the +cheapest thing to measure and the most shareable thing to publish. Two live +candidates, both checked 2026-08-27: + +- **`caveman`** — 51,690 GitHub stars within two weeks of release, advertising a + 65% token cut (75–80% in ultra mode). It also carries live public + *disagreement* — *"it is not consensus that caveman is good; a lot of people + dislike its output"* (field thread, 2026-08-31) — which fires selection + triggers 1, 2 and 3 simultaneously and makes it the strongest available + subject rather than merely the most popular one. Its own documentation concedes the claim + covers **output tokens only** — input and reasoning are untouched — and that + the skill adds roughly 1–1.5k input tokens per turn. Nobody has measured the + crossover, so on short-output, many-turn workloads it may *increase* total + spend. That boundary is an unclaimed, highly shareable finding. +- **`ponytail`** — advertised −54% code, −22% tokens, −20% cost, −27% time. A + JetBrains benchmark measured −15% code, −10.3% cost, −11% time: a quarter to a + half of the claim, with the bootstrap interval on the median just touching + zero, and savings concentrated on large tasks (−31% on 300+ line work, near + zero on minimal ones). A published prior measurement to build on or contest, + and a textbook demonstration of why a result belongs to a task class and a + model-harness pair rather than to a single headline percentage. + +This is also the answer to "the next model makes measurement irrelevant": a +better model does not fix a broken artifact it was instructed to trust, and a +declared percentage that holds only above 300 lines is a fact about the artifact, +not about the model. + +**Alternative first subject:** `find-skills` (Vercel Labs), 3.1M installs and 29.7K +stars on skills.sh, whose own description states that its quality criteria for +recommending other skills are install count above 1K, source reputation, and +GitHub stars. The most-installed skill whose job is finding good skills ranks by +popularity. It also carries a Snyk *Warn* alongside two passing audits, which +nothing in the ecosystem reconciles, because a security audit is not evidence of +function. Selection still runs through +[`../maintainer documentation: measurement-publication-playbook.md`](../maintainer documentation: measurement-publication-playbook.md), +including author contact before publication. + ### Why 16.3–16.5 cannot be pulled forward Definitional, not a priority call. `16.4`: *"two processes under one operator are @@ -112,6 +229,33 @@ entry point. - **A phase closes on its externally visible effect, not on its merge.** Step 6 is the clearest case. +- **Closing a phase is not retiring its plan, and from 15.15 on it usually must + not be.** Phases 15.10–15.12 were retired because their gate assertions sit in + `_CONTRACT_FREEZE` and declare no evidence contract, so deleting the plan cost + nothing. That stopped being true at 15.15. A plan with a declared evidence + contract is load-bearing: `phase_integrity.py` requires `phase["plan"]` to + exist, `evidence_contract.validate()` rejects an `evidence_contracts` block + whose phase has no entry, and the mutation suite indexes the shipped policy by + phase id. Deleting such a plan therefore deletes the contract, the mutation + tests that prove the auditor fires, and — because + `server_authoritative_request_fields` is only checked for *activated* phases — + the guard on live code that nobody is going to re-add. Retire a plan only when + its phase entry can be removed without removing a check. Otherwise mark the + criteria `- [x]` with a dated settlement note and say **Done** here. + [`16.1.1`](phase-16.1.1-a-gate-outlives-its-plan.md) replaces this rule with + a mechanism — the plan moves to `maintainer documentation: phases/` under its own number, + whatever guarded live code is re-homed as a standing entry first, and + `docs_integrity` refuses a retirement that is only half done. Until it lands, + this rule stands. +- **A phase entry demands at most five assertions.** Above that it splits into + numbered children that seal independently. `16.1` cost seven pull requests + and 14,443 inserted lines because nine assertions in one gate cannot be + landed in pieces; the argument and the review date are in + [`16.1.2`](phase-16.1.2-a-plan-is-one-pull-request.md). +- **A number is spent once**, whether or not it ever carried a gate, and a + retired number keeps a document in `maintainer documentation: phases/` where a check can + see it. The pre-Phase-15 blocks were deleted on 2026-08-18 precisely because + `15`, `16` and `17` had been reused for phases meaning something else. - Every critical-path subphase adds a named builtin scenario and is incomplete until it passes [the real-agent proving-ground gate](agent-proving-ground-phase-gate.md) against the locally running API. **A replayed payload is not a live run** — @@ -145,6 +289,19 @@ entry point. validate itself fails the coalition-wealth test. - Skills first. MCP follows strict safe probes. Hugging Face is metadata-first. - No phase may require Logion to own a GPU fleet. +- **The answer is free and install-free; the CLI answers a different question.** + Never gate a public answer behind installing the CLI. The CLI's pitch is the + user's own inventory, not more of the public answer. See + [`public-answer-surface.md`](public-answer-surface.md). +- **Rank by disclosed properties, never by an aggregate quality score.** This + binds hardest where Logion measures evaluators: a single "eval quality score" + makes Logion the terminal authority, which contradicts `16.5` ("authority is + local, issuer-aware, and policy-versioned"). Publish whether an eval has a + control, a statistical test, a held-out set, a contamination check, and whether + its reference solution passes its own benchmark — and let the consumer weigh. +- **A published result belongs to a model-harness pair, not to a model.** Record + harness id+version and model id+version as closed fields, and never compare two + results across differing pairs without saying so. - **No public claim can be stronger than its underlying evidence and local authority policy.** @@ -157,22 +314,42 @@ entry point. | Phase | # | Outcome | | --- | --- | --- | -| [`15.10`](phase-15.10-native-acquisition-artifact-delivery-and-inventory.md) | done | Hosted artifact downloads plus `npx skills`, `npx plugins`, `hf` acquisition/reconciliation | -| [`15.10.1`](phase-15.10.1-deepseek-harness-adapter-and-logion-dsh-plugin.md) | — | DeepSeek Harness plugins; **distribution experiment**, run after step 6 | -| [`15.11`](phase-15.11-native-use-observation-linked-feedback-and-reviews.md) | 1 | Observe native usage; feedback linked to the exact resource | -| [`15.11.1`](phase-15.11.1-publisher-integrated-consented-observation.md) | — | Publisher-shipped consented projections; **deferred**, see above | -| [`15.12`](phase-15.12-ai-catalog-publication-and-ard-discovery.md) | 3 | AI Catalog publication + ARD discovery — **in progress** on `feat/phase-15.12-ai-catalog-ard` | +| `15.10` | done | Hosted artifact downloads plus `npx skills`, `npx plugins`, `hf` acquisition/reconciliation. Plan retired 2026-08-27; shipped shape in [`../maintainer documentation: native-acquisition-and-inventory.md`](../maintainer documentation: native-acquisition-and-inventory.md) | +| `15.10.1` | — | DeepSeek Harness plugins: built and dogfooded against real `dsh`, plugin **unpublished**. **Distribution experiment**, run after step 6. Plan retired 2026-08-27; shipped shape in [`../maintainer documentation: native-acquisition-and-inventory.md`](../maintainer documentation: native-acquisition-and-inventory.md) | +| `15.11` | done | Observe native usage; feedback linked to the exact resource. Plan retired 2026-08-27; see [`../maintainer documentation: native-use-observation-and-feedback.md`](../maintainer documentation: native-use-observation-and-feedback.md) | +| `15.11.1` | — | Publisher-shipped consented projections; **never built, not now**. The Analytics role it was meant to fill is step 2, which needs no consent and no client-side code. Revisit only on the trigger in [`normative-carry-overs.md`](normative-carry-overs.md#publisher-integrated-observation--designed-not-built). Plan retired 2026-08-27; full design in git history | +| `15.12` | done | AI Catalog publication + ARD discovery. Plan retired 2026-08-27; see [`../maintainer documentation: ai-catalog-and-ard-discovery.md`](../maintainer documentation: ai-catalog-and-ard-discovery.md) | | [`15.13`](phase-15.13-portable-scan-evidence.md) | — | Signed portable evidence from current scanners | | [`15.14`](phase-15.14-feedback-driven-platform-bounties.md) | — | Platform-funded improvements from attributed usage | -| [`15.14.1`](phase-15.14.1-local-multi-agent-first-node-foundation.md) | 4 | Isolated founder-operated roles on one MacBook | -| [`15.15`](phase-15.15-isolated-first-runner-node.md) | 4 | First isolated CPU runner | +| [`15.14.1`](phase-15.14.1-local-multi-agent-first-node-foundation.md) | done | Isolated founder-operated roles on one MacBook. Sealed gate `artifacts/phase-gates/phase-15.14.1.json`. Plan kept: it is the canonical plan its policy entry binds to | +| [`15.15`](phase-15.15-isolated-first-runner-node.md) | done | First isolated CPU runner. Sealed gate `artifacts/phase-gates/phase-15.15.json`. Plan kept: it is the canonical plan behind the runner receipt/sandbox evidence contract | | [`15.16`](phase-15.16-first-party-resource-dogfood-loop.md) | — | Recurring acquire → use → evidence → bounty → rerun loop | | [`15.17`](phase-15.17-aktp-evidence-and-improvement-feed-v0.md) | 7 | AKTP v0 as an ARD-linked evidence/improvement feed | ### Phase 16 — verification [umbrella](phase-16-distributed-evaluation-and-independent-verification.md) -**Step 5:** [`16.1`](phase-16.1-eval-contract-and-reference-runner.md), +**Step 5:** [`16.1`](phase-16.1-eval-contract-and-reference-runner.md). +**Step 5a:** [`16.1.1`](phase-16.1.1-a-gate-outlives-its-plan.md), +[`16.1.2`](phase-16.1.2-a-plan-is-one-pull-request.md), +[`16.1.3`](phase-16.1.3-the-documentation-is-read-out-of-the-code.md) — the +plan cadence, numbered but deliberately ungated: they change policy and +generators, have no customer effect to seal, and prove themselves with `test:` +markers. + +**Parked, off the critical path, both written 2026-09-08 out of one +measurement:** [`16.1.4`](phase-16.1.4-a-published-entry-resolves.md) — a third +of our own AI Catalog is a dead pointer, the exported contract declares no +security scheme, and ten money-moving POSTs have no idempotency key. No +precondition; it is a defect plan and every finding in it is reproducible from +a terminal today. [`16.1.5`](phase-16.1.5-what-is-installed-here-has-a-version.md) +— one of the fourteen skills installed on the operator's machine carries a +version, it is Logion's own, and it is stale. No technical precondition either: +it is the N=1 the thesis claims, run on the machines the work actually happens +on, and what it competes with is time, not a dependency. Neither is required +for 0.2. + +**Step 5b:** [`16.2`](phase-16.2-typed-evaluators-and-skill-reference-evaluator.md). `16.2` is also what satisfies "must work for more than MCP" — the `Evaluator` protocol is typed per resource, so skills, plugins, MCP servers and models all @@ -198,7 +375,9 @@ in the evaluation layer, not the observation layer.** path except [`17.3`](phase-17.3-resource-claims-and-commercial-rails.md) at step 9 and [`17.6`](phase-17.6-public-narrative-and-landing-truth-pass.md), **pulled forward into the 0.2 release gate** because the landing still describes a -marketplace. +marketplace. `17.6` is now executed *through* +[`public-answer-surface.md`](public-answer-surface.md) at step 5c: the landing +rewrite and the evidence report page are one deliverable, built report-first. Remaining: [`17.1`](phase-17.1-ai-catalog-ard-aktp-conformance-and-upstream-proposals.md), [`17.2`](phase-17.2-independent-node-federation.md), diff --git a/plans/normative-carry-overs.md b/plans/normative-carry-overs.md index a4c272bf..46180f77 100644 --- a/plans/normative-carry-overs.md +++ b/plans/normative-carry-overs.md @@ -3,9 +3,10 @@ # Normative carry-overs Inherited implementation contracts that survived the retirement of their -original plan files. These are **binding** on 15.10/15.11/15.11.1 and are not -waived by any resequencing. Read this file when implementing identity, -acquisition, reconciliation, or observation — not otherwise. +original plan files — 15.10, 15.10.1, and 15.11 among them. These are +**binding** and are not waived by any resequencing or by a plan being deleted. +Read this file when implementing identity, acquisition, reconciliation, or +observation — not otherwise. Split out of `next-steps.md` on 2026-08-17 so the execution order stays short. The text below is unchanged. @@ -21,13 +22,26 @@ remaining normative. `publish-from-repo` command — repo publishing stops at the source link; - the platform-bounty admin lane is API-only (no SDK resource, no `logion admin bounties` CLI subgroup); -- `cli/_observation.py` ships the local observation envelope/spool writer as a - library with no production command wired to it; - `resources acquire` behavior is owned by 15.10; - the normative `scope_id`/`installation_id` HMAC and cryptographic publisher-signature verification are not implemented; - the indexer has no `lobehub` adapter (`skillsmp` and `smithery` shipped - instead), and the `hermes_docs` adapter has no seed entry. + instead), and the `hermes_docs` adapter has no seed entry; +- **neither 15.10 nor 15.10.1 has a sealed real-agent phase gate.** Both plans + required one and both plans are now retired. The scenarios exist + (`acquisition_install_scope_safety`, + `dsh_plugin_discovery_install_and_reconcile`) and both phases have real + dogfood records against live managers, but `logion/artifacts/phase-gates/` + holds only `phase-15.11`, `phase-15.12`, and `phase-15.14.1`, and there is no + run report for either scenario. The obligation survives the plan; +- **the Logion dsh plugin is unpublished.** `plugins/dsh-plugin/` + (`@logionsh/dsh-plugin`) sits at `0.1.0` with no `dsh-plugin-v*` tag, so + 15.10.1's "installs through dsh's native flow from its public distribution + point" criterion is unmet. Release procedure is in `logion/RELEASING.md`; the + roadmap sequences this as a distribution experiment after step 6. + +Shipped shape for acquisition, inventory, reconciliation, and the dsh path is +[`../maintainer documentation: native-acquisition-and-inventory.md`](../maintainer documentation: native-acquisition-and-inventory.md). ## Local installation identity @@ -120,41 +134,129 @@ minimum-disclosure proposal; `auto` = only the separately documented narrow receipt class. Ratings, prose, and raw task data always need separate consent. **An observation is not a rating.** -### Envelope conflict — resolve before 0.2 +### Envelope conflict — resolved 2026-08-27 + +Two envelopes were normative for the same record: the live `UsageObservation` +spool schema, and the richer `cli/_observation.py` envelope (task class, +outcome, ordered timestamps, integration version), which had no production +caller. **Settled by adoption and deletion**: `cli/usage/observations.py` is the +single normative envelope, carrying the richer fields, and `cli/_observation.py` +is deleted. The spool rejects a record whose `integration_version` does not +match, so a stale hook cannot write the old shape. + +### Envelope fields — decided 2026-08-27, unbuilt + +Recorded here rather than in `15.11` because that phase is closed and its plan +file is being retired; this contract has to outlive it. It is not waived by any +resequencing, and it belongs to whichever phase next touches the envelope. + +The shipped envelope records `harness` as a bare lowercase slug with no version, +and records nothing about the model. Both are gaps. The sequencing argument is +the same for both and is about schema, not product: the spool rejects any record +whose `integration_version` does not match, so adding a field later is a version +bump that invalidates every installed hook. **The cheapest moment to change a +strictly-versioned envelope is while the installed base is approximately zero** — +true now, false after 0.2. Gate item in [`release-0.2.md`](release-0.2.md). + +**Harness version — add.** `claude-code` in March and `claude-code` in August are +not the same harness, and the difference can move a result more than the model +does ([arXiv:2605.23950](https://arxiv.org/abs/2605.23950)). Without it, the +"harness coverage" that every published `n` is required to carry does not name +anything specific. Zero tokens, no new data category, no new consent question. + +**Model slug — add, as a closed allowlist only.** Without it a field cohort +silently aggregates across Opus, Haiku and gpt-5.4-mini, which are not one +population; "fails only on a small model" is exactly the finding a publisher can +act on and a bounty can fix. Conditions, all mandatory: + +- a **closed allowlist of coarse slugs**, never free text, never a private + endpoint or deployment name; +- **declared per adapter**, like every other capability here — most adapters will + honestly report they cannot see it, and that refusal is the correct outcome + rather than an inference; +- same consent axis as the rest of the envelope, and suppressed in aggregates + below the minimum cohort, because a model slug at low `n` is mildly + identifying; +- if it cannot be a closed enum, it does not ship. -Two envelopes are currently normative for the same record: the live -`UsageObservation` spool schema, and the richer `cli/_observation.py` envelope -described above (task class, outcome, ordered timestamps, integration version), -which has no production caller. 15.11.1 either adopts one or the other is -deleted. Both cannot stay authoritative. This is a gate item in -[`release-0.2.md`](release-0.2.md). +This does **not** reopen the `model context` prohibition above, and the +distinction is load-bearing: the prohibition covers prompts, system text and +conversation state. A coarse model slug is an environment label of the same kind +as `harness`. `15.11.1`'s instrumentation profile keeps `model_context` in its +`excluded` list unchanged. -## Publisher-integrated observation — capability correction +The slug is what makes **model-harness pair** evidence possible, which is the one +form of model ranking that is not commodity — everyone publishes model rankings, +nobody publishes pair rankings, and `harness` already being first-class is why +Logion can. -[`15.11.1`](phase-15.11.1-publisher-integrated-consented-observation.md) -declares a static Agent Skill without hooks as -`publisher_observation_unsupported`. **That is out of date.** Claude Code -supports a `hooks` field in `SKILL.md` frontmatter; hooks declared there are -registered when the skill is invoked and persist for the rest of the session, -with an `once: true` option. Verified against the current published skills and -hooks references, 2026-08-17. +**Measured token counts — refuse.** No hook payload carries per-resource token +attribution, so any number here would be inferred, and an inferred number is what +this file forbids everywhere else. Token measurement belongs to `16.2`, where the +runner controls the loop. The existing `token_efficiency` score stays a declared +opinion, not a measurement. -Two constraints that must be written into the phase before implementation: +## Publisher-integrated observation — designed, not built -1. `hooks` is a Claude Code extension, **not** part of the Agent Skills spec. +The publisher path — a resource owner instruments their own artifact so a user +who never installs the Logion CLI can consent to and emit a narrow usage receipt +— was designed in full and **never implemented**. It is not on any queue. The +full design is in git history at +`plans/phase-15.11.1-publisher-integrated-consented-observation.md` (retired +2026-08-27); its capability tiers and per-client coverage are already reflected +in the shipped typed refusals documented in +[`../maintainer documentation: native-use-observation-and-feedback.md`](../maintainer documentation: native-use-observation-and-feedback.md). + +**Why it is not built — the reasoning, so it is not re-derived wrongly.** It is +tempting to call this the Google Analytics move: the publisher gets their own +numbers, Logion gets visibility outside its own realm. That role is real, but +this is the expensive way to fill it. + +1. **The cheap version is already on the critical path.** Cross-hub install + reconciliation (step 2) gives a publisher something no hub can give them — + presence and version coverage of their artifact across every indexed hub — + and, as [`next-steps.md`](next-steps.md) puts it, de-fragmentation needs no + consent and no client-side code. Same trade, none of the machinery. +2. **The mechanism's own premise limits its reach.** It works only for a client + whose hook contract is pinned to an exact release with a recorded payload + fixture. Today that is Claude Code and Codex; everything else resolves to + `unsupported`, and the Hermes fixture the design treated as gate-required was + never recorded. "Instrument once, reach everyone" was never true. +3. **The distribution problem it existed to solve has a cheaper answer.** Its + job was reaching people who never install the Logion CLI. The public answer + surface ([`public-answer-surface.md`](public-answer-surface.md)) does that + with no consent, no client-side code, and no publisher adoption. + +**The trigger that would make it right.** Today, instrumenting is a *push* at +publishers who have never heard of Logion, which is why it reads as noise. +Revisit when it becomes a *pull*: **a publisher who already holds a Logion +report about their own artifact asks to see it continuously, for their own +users.** At that point the adoption problem is already solved and the mechanism +is worth its cost. That demand comes from step 8 outreach, not from step 1 +supply — so nothing before step 8 should schedule this. + +One constraint outlives it, because it binds any future harness integration and +is recorded nowhere else. + +### Claude Code `SKILL.md` frontmatter hooks + +Claude Code supports a `hooks` field in `SKILL.md` frontmatter; hooks declared +there register when the skill is invoked and persist for the session, with an +`once: true` option (verified 2026-08-17). Two constraints: + +1. `hooks` is a **Claude Code extension, not part of the Agent Skills spec**. The spec permits `name`, `description`, `license`, `compatibility`, `metadata`, `allowed-tools`, and an unknown key is a **hard packaging error** for claude.ai upload, the Skills API, and `package_skill.py`. Instrumenting a - skill this way costs the publisher those distribution paths. `metadata` is - the only in-spec, portable carrier for an instrumentation-profile reference, - and it is declarative only — it cannot execute. + skill this way costs the publisher those distribution paths. `metadata` is the + only in-spec, portable carrier for a reference of this kind, and it is + declarative only — **metadata a manager tolerates is not metadata a harness + executes.** 2. There is **no consent prompt before a skill-registered hook runs a command**. - The disclosure gate is entirely Logion's responsibility, and a - network-calling hook inside a third party's artifact is the single largest - reputational risk in the product. Consent must be a visible, verifiable - badge, never fine print. The existing fail-open and exact-disclosure - requirements in that plan are the floor, not the ceiling. - -Because of (1), prefer plugin and MCP projections first, where the hook is a -native, expected mechanism. Skill-frontmatter instrumentation ships only where a + The disclosure gate is entirely Logion's responsibility, and a network-calling + hook inside a third party's artifact is the single largest reputational risk + in the product. Consent must be a visible, verifiable badge, never fine print. + +Because of (1), prefer plugin and MCP surfaces first, where a hook is a native, +expected mechanism. Skill-frontmatter instrumentation ships only where a publisher explicitly accepts the packaging trade-off. diff --git a/plans/phase-15.10-native-acquisition-artifact-delivery-and-inventory.md b/plans/phase-15.10-native-acquisition-artifact-delivery-and-inventory.md deleted file mode 100644 index 4f58bf23..00000000 --- a/plans/phase-15.10-native-acquisition-artifact-delivery-and-inventory.md +++ /dev/null @@ -1,475 +0,0 @@ - - -# Phase 15.10 — Native acquisition, artifact delivery, and local inventory - -> **Implementation status (2026-07-30): not shipped.** This document is the -> normative future contract and acceptance gate. The current CLI can inspect -> and dry-run only a local source skeleton; the public API does not yet return -> distribution URLs, permissions, acquisition plans, or artifact downloads. -> Therefore non-dry-run acquisition is blocked rather than presented as an -> executable install. -> -> **Dogfood starts here — Level 1 (real acquisition):** the implementing agent must discover a resource in Logion, acquire its actual artifact, use it from a normal agent workflow, and prove that Logion can reconcile the installed copy. -> **After this phase:** a user may acquire through Logion or keep using `npx skills`, `npx plugins`, or `hf download`; Logion records one canonical local inventory either way. -> **Honesty boundary:** acquisition and installation prove possession of an artifact, not usage, usefulness, safety, or entitlement to paid content. - -## Why this phase moved before ARD, evidence, and runners - -The product cannot dogfood by inspecting metadata while the implementing agent cannot download the selected artifact. More importantly, Logion must not demand that users abandon native ecosystems: - -- skills use `npx skills add|use`; -- agent plugins use `npx plugins add`; -- Hugging Face resources use `hf download --revision`; -- Logion-hosted Course/capability artifacts need a real authenticated download path. - -Logion's job is to resolve identity, display policy and evidence, produce an acquisition plan, verify the resulting artifact, and remember where it is installed. It is not to replace every upstream package manager. - -## Mandatory dogfood prompt for the implementing agent - -This prompt is a release gate and runs after the implementation is test-green: - -```text -You are implementing Phase 15.10. Dogfood Logion's actual artifact path. - -1. Run `logion recall search "artifact acquisition package manager interop" --limit 5`. -2. On LOW/NONE, run: - `logion listings search --query "artifact acquisition package manager interop" - --include-indexed --limit 5 --json`. -3. Select one published/free Course resource that has a Logion-hosted bundle and one - externally indexed skill that recommends `npx skills`. -4. Inspect each resource, exact version, distribution channel, permissions, license, - expected bytes, digest, source, and acquisition command. -5. Ask for explicit approval before either installation. For both resources run: - `logion resources acquire RESOURCE_ID --version VERSION_ID --scope repo-root - --channel auto --dry-run --json`. -6. Acquire the hosted Course through `logion_bundle`. Acquire the external skill by - allowing Logion to delegate to the displayed `npx skills add ...` command. -7. Run `logion resources inventory --scope repo-root --json` and - `logion resources reconcile --scope repo-root --json`. -8. Use both resources in harmless, bounded implementation tasks. Confirm the harness - loads the same files/digests recorded by inventory. -9. Save `artifacts/dogfood/phase-15.10.md` with resource/version IDs, acquisition plans, - approval, exact executed commands, upstream locator/revision, installed paths, - digests, reconciliation outcome, and product friction found. -10. Until 15.11 exists, submit `logion courses report-usage` only for the hosted Course - if it was actually used and the current review eligibility permits it. Do not - invent a review for the ownerless indexed skill. -``` - -If this phase cannot download a real hosted Course artifact, it does not pass. - -## Dependencies - -- 15.9.1 harness resource scope and observation contract. -- 15.3 package maps and source provenance. -- 15.6 mirrored indexed bundles and source attribution. -- 15.9 generic `Resource`, `ResourceVersion`, `ResourceSource`, and compatibility projections. -- Existing `CourseAsset`, object storage, entitlements, `_local_state.py`, install finalization, harness projection, and CLI confirmation helpers. - -## Upstream contracts to pin before coding - -- `skills` CLI source/README and release: -- skills.sh CLI/telemetry documentation: -- Vercel plugin entry surface and current plugin repository/spec references: and -- Hugging Face `hf download` reference, including `--revision`, `--local-dir`, cache, token, and `--dry-run`: -- Codex skill discovery and repository/user/admin scopes: -- Claude Code project, personal, and plugin scopes: -- Hermes skills and profile isolation: and -- Pi project/user skill discovery and `.agents/skills` compatibility: - -Record exact tested versions and fixture provenance in the PR. These tools can change; adapters must fail closed on an unsupported state format/version rather than silently misattribute. - -## Product contract - -### One resource, multiple distributions - -A `ResourceVersion` may expose zero or more immutable acquisition distributions: - -```text -logion_bundle authenticated object-store bundle controlled by Logion -npx_skills upstream Git/git-hosted Agent Skill acquired by `skills` -npx_plugins upstream agent plugin acquired by `plugins` -hf model/dataset/space revision acquired by `hf` -git exact public Git commit for unsupported native managers -manual metadata-only instruction; Logion does not execute it -``` - -The `channel` is not the resource identity. The same content may be available through multiple channels. Evidence remains attached to `ResourceVersion.content_digest`. - -### Acquisition plan - -`GET /v1/resources/{resource_id}/versions/{version_id}/acquisition-plan` -returns the server-owned resource/distribution plan. It never receives or -returns a local path, `scope_id`, or `installation_id`: - -```json -{ - "resource_id": "uuid", - "version_id": "uuid", - "distribution_id": "uuid", - "content_digest": "sha256:...", - "selected_channel": "npx_skills", - "alternatives": ["logion_bundle", "git"], - "entitlement": {"required": false, "status": "not_applicable"}, - "license": {"spdx": "MIT", "redistribution_allowed": true}, - "expected": {"bytes": 1234, "files": 4}, - "native": { - "tool": "skills", - "tested_version": "x.y.z", - "argv": ["npx", "skills@x.y.z", "add", "owner/repo", "--skill", "name"], - "upstream_locator": "https://github.com/owner/repo", - "revision": "40-char-commit" - }, - "integrity": {"algorithm": "sha256", "digest": "..."}, - "permissions": {"network": false, "tools": [], "secrets": []}, - "warnings": [] -} -``` - -The public CLI validates that response and combines it locally with the 15.9.1 -harness-scope resolver. The resulting local acquisition plan, including its -zero-write dry-run serialization, contains: - -```json -{ - "harness": "codex", - "requested_scope": {"kind": "repo-root"}, - "resolved_target": { - "scope_kind": "repo-root", - "scope_id": "opaque-profile-node-scoped-id", - "relative_path": ".agents/skills/name", - "precedence": 30 - } -} -``` - -The CLI may display the resolved absolute path interactively, but it must not -send that path to the acquisition-plan endpoint or persist it in remote -telemetry. Dry-run does not contain `installation_id` or -`native_receipt_digest`; those fields exist only after validated native evidence -is available. - -`native.argv` is a display/execution array, never a shell string. User-controlled values cannot become flags without `--`/adapter validation. The server never executes it. - -### Local acquisition receipt - -Every successful acquisition/reconciliation writes a fixed-schema local record: - -```json -{ - "schema_version": 1, - "resource_id": "uuid", - "version_id": "uuid", - "distribution_id": "uuid", - "resource_type": "agent_skill", - "content_digest": "sha256:...", - "channel": "npx_skills", - "upstream_locator": "owner/repo@commit#skill-name", - "harness": "codex", - "scope_kind": "repo-root", - "scope_id": "opaque-profile-node-scoped-id", - "installation_id": "opaque-profile-node-scoped-installation-id", - "native_receipt_digest": "sha256:64-lowercase-hex", - "native_evidence": { - "schema_version": 1, - "manager_name": "skills", - "manager_version": "x.y.z", - "receipt_id": "native-lock-or-receipt-id", - "canonical_source": "https://github.com/owner/repo", - "immutable_revision": "40-char-commit", - "content_digest": "sha256:..." - }, - "target_path": "/workspace/xpto/.agents/skills/skill-name", - "relative_target_path": ".agents/skills/skill-name", - "installed_paths": [".agents/skills/skill-name"], - "projection_paths": [".claude/skills/skill-name"], - "acquired_at": "RFC3339", - "verified_at": "RFC3339", - "verification": "exact|source_revision|unverified" -} -``` - -No telemetry or review is sent by creating this record. `native_receipt_digest` -is recomputed from the RFC 8785 canonical JSON bytes of `native_evidence` before -the receipt is accepted; a mismatch fails closed. `native_evidence.manager_name` -and `native_evidence.manager_version` are the single authoritative manager -identity and produce -the 15.9.1 `native_manager` value `@`; the receipt -must not duplicate those fields at top level. `scope_id` and the -installation identity use the profile/node-scoped, domain-separated HMAC -contract from 15.9.1. A plain SHA-256 of `target_path`/repository path is -forbidden. Absolute paths remain local and are never copied into API payloads. -For `verification: exact`, `native_evidence.content_digest` must equal the -top-level immutable resource `content_digest`, and its canonical source/revision -must match the selected distribution. A mismatch is retained only as unlinked -inventory evidence and cannot mint `installation_id`. -`relative_target_path` is the single canonical primary target used by the -15.9.1 installation HMAC. `target_path` is its local absolute rendering; -`installed_paths` and `projection_paths` are lifecycle evidence and do not alter -that installation identity. - -## Database and API implementation - -### Migration - -Add `resource_distributions`: - -```text -id UUID PK -resource_version_id UUID FK resource_versions(id) ON DELETE CASCADE -channel VARCHAR(32) NOT NULL -locator TEXT NOT NULL -upstream_revision TEXT NULL -artifact_digest VARCHAR(80) NULL -artifact_size_bytes BIGINT NULL -media_type TEXT NULL -metadata JSONB NOT NULL DEFAULT '{}' -priority SMALLINT NOT NULL DEFAULT 100 -enabled BOOLEAN NOT NULL DEFAULT TRUE -created_at, updated_at -UNIQUE(resource_version_id, channel, locator) -``` - -Do not put ephemeral presigned URLs in this table. A Logion bundle distribution references existing immutable Course assets/object keys through metadata validated by a service. - -### `backend repository` files - -- Add `api/resources/services/build_acquisition_plan.py`. -- Add `api/resources/services/get_resource_artifact_download.py`. -- Add `api/resources/services/register_resource_distribution.py`. -- Add repository/controller/response types under existing `api/resources/`. -- Reuse `api/storage/services/s3_storage_service.py` for short-lived download URLs. -- Reuse payments entitlement services for paid Course projections. -- Extend Course finalization/publication to create/update `logion_bundle` distributions only after asset manifest and digest validation. -- Extend indexed upsert to create `npx_skills`, `git`, or mirror distributions from trustworthy source/package-map data. -- Add rate limits, object-size caps, audit events, and metrics. - -### API endpoints - -- `GET /v1/resources/{id}/versions/{version_id}/acquisition-plan` -- `POST /v1/resources/{id}/versions/{version_id}/download` for authorized Logion bundles; response is a short-lived manifest/URL, not raw bytes through FastAPI. -- Admin/internal distribution registration remains behind indexing/publication services; no arbitrary public URL registration endpoint. - -Stable errors: - -```text -resource_distribution_unavailable -resource_version_digest_missing -resource_artifact_not_redistributable -resource_entitlement_required -resource_entitlement_inactive -resource_artifact_digest_mismatch -resource_native_tool_unsupported -resource_native_tool_version_unsupported -resource_acquisition_channel_denied -``` - -## CLI and client implementation - -### Generated/public client - -- Add typed acquisition-plan and artifact-download methods to `logion/packages/client/src/logion/v1/_resources/resources.py`. -- Regenerate OpenAPI operations/types; never hand-edit generated files. - -### New CLI package - -Add `logion/packages/cli/cli/commands/resources/`: - -```text -parser.py -handlers.py -acquire.py -inventory.py -reconcile.py -distributions.py -adapters/base.py -adapters/logion_bundle.py -adapters/npx_skills.py -adapters/npx_plugins.py -adapters/hf.py -``` - -Commands: - -```bash -logion resources distributions RESOURCE_ID --version VERSION_ID --json -logion resources acquire RESOURCE_ID --version VERSION_ID \ - --harness codex|claude|hermes|pi|opencode \ - --scope repo-current|repo-parent|repo-root|user|admin|custom \ - --channel auto|logion_bundle|npx_skills|npx_plugins|hf \ - --dry-run --json -logion resources inventory [--harness HARNESS] [--scope SCOPE|all] [--json] -logion resources reconcile [--from skills|plugins|hf|logion|all] \ - [--harness HARNESS|all] [--scope SCOPE|all] [--dry-run] [--json] -``` - -`acquire` execution order: - -1. Fetch and validate plan. -2. Detect the harness and resolve scope from the current working directory. - Inside a Git repository the default is the root of the nearest containing - Git worktree; there is no silent fallback to `user`. `system` is - inventory-only. -3. Display price/entitlement, license, bytes, digest, native tool/version, permissions, paths, and exact argv. -4. `--dry-run` performs no download, package-manager execution, config write, or inventory mutation. -5. Ask explicit confirmation unless caller already supplied the normal approved non-interactive flag. -6. Execute adapter without a shell. -7. Discover actual installed paths/output using adapter-specific state. -8. Verify revision/digest to the strongest available level. -9. Write inventory atomically. -10. Project to the selected harness's native directory using native manager - behavior first; Logion must not create duplicate copies. - -Both directions are normative: Logion must install a catalog resource into the -requested native harness scope, and it must reconcile a resource installed -directly through a native manager without moving or reinstalling it. - -### Logion bundle adapter - -- Download manifest and files to a temporary directory. -- Verify every file size/digest and aggregate content digest before installation. -- Call existing install/finalization libraries rather than duplicating `_install_helpers.py`. -- Record Course/version/resource provenance in the manifest. -- Delete partial/temp data on failure; keep resumable cache only when content-addressed. - -### `npx skills` adapter - -- Verify `node`/`npx` and supported `skills` version. -- Use documented `npx skills add --skill ` arguments, - repository scope by default. Never interpret “project” as user-global. -- Do not pass `-y` unless the outer Logion approval has already been obtained and the full argv was displayed. -- Read `skills-lock.json`, canonical skill directory, symlinks/copies, and manager output after completion. -- Preserve upstream lock entries and files. Logion adds its own local inventory; it does not rewrite `skills-lock.json` unless compatibility requires a bit-exact, tested field update. -- Also reconcile resources installed before Logion by reading the lockfile and exact source metadata. - -### `npx plugins` adapter - -- Treat a plugin as a distinct `agent_plugin` resource; bundled skills remain child/source relationships, not duplicate Course ownership. -- Use the plugin manager's official manifest/state and supported agent projections. -- Never infer a plugin ID from directory basename alone. -- Reconcile existing plugin installs without rewriting their manager files. - -### `hf` adapter - -- Produce `hf download REPO_ID --revision COMMIT` or an `hf://...@COMMIT` locator. -- Default acquisition plan is metadata/files explicitly required by the consuming eval/workflow. Never download all model weights from index/search. -- Run `hf download --dry-run` as part of Logion dry-run when available. -- Read Hub cache revision/snapshot metadata and verify commit/file metadata. -- Tokens stay in the native `hf` credential path and are never copied into Logion inventory. - -## Reconciliation and identity rules - -Attribution priority: - -1. exact Logion resource/version marker; -2. exact native lock/manifest source plus immutable revision; -3. exact content digest; -4. canonical source plus verified subpath/name; -5. unresolved. - -Never fuzzy-link by display name. Multiple candidates produce `ambiguous` with candidate IDs and no attribution. - -`reconcile` may: - -- create/update local inventory; -- mark missing/drifted installations; -- improve verification when a digest becomes available. - -It may not: - -- upload telemetry; -- submit feedback/review; -- delete native manager state; -- upgrade/downgrade artifacts; -- claim ownership or entitlement. - -## Security and privacy - -- All archives use traversal, symlink, decompression-ratio, file-count, type, and size defenses already used by Course upload/install paths. -- Native command invocation uses argv, sanitized environment, cwd pinned to project, timeout, output cap, and no secret logging. -- The acquisition plan is untrusted server/source metadata until client validation. -- Logion-hosted paid artifacts require entitlement; external public copies do not silently mint entitlement. -- Local inventory stores no prompts, tool inputs, user identity, tokens, or repository contents. - -## Tests - -### Backend - -- Hosted free/paid/expired entitlement plans and downloads. -- Presigned URL expiry, object/digest mismatch, non-redistributable license, disabled distribution. -- Course publication creates one idempotent bundle distribution. -- Indexed source creates exact native distribution; ambiguous/missing revision is quarantined or manual-only. -- OpenAPI and generated-client contract. - -### CLI - -- Plan rendering and `--dry-run` zero-write/zero-exec. -- Fake executable adapters assert exact argv and shell is never used. -- Hosted bundle happy path, partial download, digest mismatch, traversal, resume/cleanup. -- Recorded `skills` fixtures: repository/user, symlink/copy, multiple skills, - lock drift, pre-existing installation, unknown manager version. -- Recorded `plugins` manifest/state fixtures and unsupported agent. -- Recorded `hf download --dry-run`/cache fixtures, exact revision, gated token non-leak, oversized plan. -- Ambiguous identity never links; second reconcile is zero-change. -- Exact discovery/scope fixtures for Codex, Claude, Hermes, and Pi as specified - in 15.9.1, including precedence and unsupported-scope failures. -- Install the same resource in fixture repositories `xpto` and `acme`; verify - distinct receipts and no repository/user-scope leakage. -- Launch a fresh real harness session and assert native discovery of the exact - installed version. -- Scope isolation and existing `skills install/list/inspect/update/prune` regressions. - -## Rollout - -1. Hosted free Course bundles. -2. Existing indexed skill reconciliation without executing `npx`. -3. Delegated `npx skills` acquisition. -4. Plugin reconciliation/acquisition. -5. HF metadata/selective-file acquisition. -6. Paid hosted bundles after entitlement and red-team tests. - -Feature flags exist per channel. Metrics include planned/started/succeeded/failed acquisition, bytes, verification level, reconcile matched/ambiguous/drifted, native tool/version, and channel—but no user project paths. - -## Mandatory proving-ground scenario - -Follow [the common real-agent gate](agent-proving-ground-phase-gate.md). Add -`builtin:phase_15_10_native_acquisition`. - -- **Actors/fixtures:** a fresh `buyer` and separate `operator`. - `make proving-ground-seed SCENARIO=phase_15_10` publishes a small hosted - bundle, repositories `xpto` and `acme`, isolated user homes, and local Git - fixtures accepted by the real `npx skills` and `npx plugins` CLIs. No fake - `logion`, `npx`, harness, or adapter executable. -- **Customer prompt:** “I need a lightweight code-review capability. Search - Logion, install the best suitable option for Codex in repository xpto—not - globally—show what was installed, then reconcile anything already present. - Start a fresh Codex session and prove it discovers the capability. Do not call - Logion's HTTP API directly.” -- **Flow:** acquire the hosted resource, install the indexed skill through real - `npx skills`, rerun safely, then use public inventory/reconcile commands. -- **Assertions to implement:** `api.resource_acquisition_exists`, - `api.resource_distribution_selected`, `api.native_install_reconciled`, - `files.inventory_receipt_matches`, - `files.installed_artifact_digest_matches`, and - `api.acquisition_idempotent`; require no 500s. -- **Negative case/evidence:** an advertised digest mismatch fails closed and - creates no success receipt or entitlement. Retain native tool version/output, - distribution, receipt/artifact digests, and proof of zero duplicate state. - -## Acceptance criteria - -- [ ] A published free Course discovered through Logion downloads and installs without a manually supplied `--source` directory. -- [ ] A resource installed directly with `npx skills add` is reconciled to the exact Logion `ResourceVersion` without reinstalling it. -- [ ] Codex, Claude, Hermes, and Pi adapters declare and test native locations, - scope precedence, observation capability, and failure behavior. -- [ ] Installing in repository `xpto` creates nothing in the user scope or - another repository; a fresh target harness discovers the exact version. -- [ ] A Vercel plugin installed with `npx plugins add` and an HF revision downloaded with `hf download --revision` can appear in local inventory without Logion owning their download path. -- [ ] Every acquisition dry-run is zero-write and shows exact channel, revision, bytes, permissions, argv, and verification expectation. -- [ ] No fuzzy name attribution, shell invocation, ambient credential copy, or hidden telemetry occurs. -- [ ] Existing Course/skills CLI and entitlement behavior remain compatible. -- [ ] The mandatory dogfood artifact contains one real hosted acquisition and one real native-manager acquisition. - -## Out of scope - -Usage observation, telemetry upload, feedback/reviews (15.11), ARD (15.12), signed scan evidence (15.13), MCP execution, arbitrary model evaluation, automatic upgrades, and funding decisions. diff --git a/plans/phase-15.10.1-deepseek-harness-adapter-and-logion-dsh-plugin.md b/plans/phase-15.10.1-deepseek-harness-adapter-and-logion-dsh-plugin.md deleted file mode 100644 index f5da0ff1..00000000 --- a/plans/phase-15.10.1-deepseek-harness-adapter-and-logion-dsh-plugin.md +++ /dev/null @@ -1,396 +0,0 @@ - - -# Phase 15.10.1 — DeepSeek Harness (dsh) native adapter, plugin acquisition channel, and Logion dsh plugin - -> **Implementation status (2026-08-14): not shipped.** This document is the -> normative future contract and acceptance gate. It is the first application of -> the 15.9.1 harness contract and the 15.10 acquisition contract to an -> ecosystem that launched after both were written. -> -> **Dogfood — Level 1.1 (new-ecosystem acquisition):** the implementing agent -> must install the real `dsh` CLI, install the Logion dsh plugin through dsh's -> own native flow, discover an indexed dsh plugin through Logion, acquire it, -> reconcile a second plugin that was installed directly with dsh, and prove a -> fresh dsh session loads what the inventory says it loads. -> **After this phase:** a dsh user can see the Logion plugin, install it, and -> use the central Logion loop — search, evidence, acquire, inventory, -> reconcile — without leaving their harness or abandoning dsh's native -> plugin management. -> **Honesty boundary:** acquisition proves possession, not usage, usefulness, -> safety, or quality. Every evidence surface shown inside dsh is first-party -> ("Logion observed"); nothing in this phase may present or imply "network -> verified", which requires Phase 16. - -## Why this phase exists - -DeepSeek Harness (`dsh`) shipped 2026-08-13 as an MIT developer preview built -on the Cordis plugin kernel: every capability — models, tools, skills, -sessions, sandboxes, storage, loops, UI — is a plugin. Distribution is -registry-less: plugins are GitHub repositories tagged with the `dsh-plugin` -topic and carry a `dsh.plugin.json` manifest. There is no marketplace, no -trust layer, and no verification model. - -That is exactly the gap Logion's wedge thesis names, at day one of the -ecosystem instead of after it hardens. Two properties make dsh a better-than- -usual fit for the existing contracts: - -- Cordis plugins declare their dependencies and provided services statically - (typed coeffects). That is a machine-readable declared-capability surface - that feeds the acquisition plan `permissions` display and, later, the - declared-vs-observed evidence lane — without inference. -- Cordis effects are revertible: unloading a plugin structurally undoes what - it installed. This makes bounded install/inspect/rollback flows cheaper. - -The integration follows the standing rule: **native managers remain native**. -Logion does not host, mirror, or replace dsh plugin distribution. It indexes, -attributes, displays evidence, plans acquisition, delegates installation to -the native flow, and remembers one canonical local inventory. - -Because dsh is a developer preview that announces compatibility-breaking -changes, the convergence kill criterion "collaboration depends indefinitely on -an unreviewed moving branch" applies. The consequences are contractual: - -- every adapter pins the exact tested `dsh` version and manifest schema and - **fails closed on any unknown format/version** rather than misattributing; -- no DeepSeek-specific protocol contract (coeffect schema in the catalog, a - Cordis-shaped receipt, a selection descriptor) is frozen in this phase; -- upstream engagement follows the ASM collaboration pattern: approach early, - prove one seam, upstream after evidence. Proposing that DeepSeek publish an - AI Catalog document is Phase 17.1 material, not this phase. - -## Mandatory dogfood prompt for the implementing agent - -This prompt is a release gate and runs after the implementation is test-green: - -```text -You are implementing Phase 15.10.1. Dogfood Logion's dsh path for real. - -1. Install the pinned tested dsh release in an isolated HOME and record the - exact version. -2. Install the Logion dsh plugin through dsh's documented native install flow - (no manual file copying). Confirm a fresh dsh session lists it. -3. From inside dsh, use the Logion plugin to search for a dsh plugin capability - and inspect one indexed result: exact version, canonical source, revision, - digest, license, declared dependencies/services, and acquisition command. -4. Run the dry-run acquisition, review the plan, then approve and acquire it - through the delegated native dsh flow into a fixture repository scope. -5. Install a second, different dsh plugin directly with dsh (bypassing Logion), - then run Logion inventory and reconcile and confirm the direct install is - attributed to the exact indexed ResourceVersion without reinstalling it. -6. Start a fresh dsh session and confirm it discovers both plugins with the - same files/digests recorded by inventory. -7. Save artifacts/dogfood/phase-15.10.1.md with dsh version, plugin manifest - digests, resource/version IDs, acquisition plans, approval, exact executed - commands, installed paths, reconciliation outcome, and product friction. -8. Do not submit any rating, review, or usage claim: observation and feedback - are 15.11. Record blockers instead. -``` - -If this phase cannot acquire a real dsh plugin through the real `dsh` flow and -reconcile a real out-of-band install, it does not pass. - -## Dependencies - -Hard prerequisites, in order: - -1. **15.10 gap closure.** The audited defects block this phase and are owned - by 15.10, not duplicated here: - - `verification: exact` must recompute the aggregate content digest - client-side and fail closed on mismatch; - - the bundle download egress must go through the guarded HTTP path - (no `urllib.urlopen`, no `file://`, explicit timeout); - - `logion resources reconcile` must read real native manager state with - `--from` sources instead of re-emitting Logion's own receipts; - - channel adapters must derive installed paths and lock entries from - manager state, never from a Logion-invented slug or an arbitrary - lockfile entry. -2. **Indexer generic resource path.** The dormant `DiscoveredResource` / - `_serialize_resource_item` pipeline must be wired and the indexer↔API type - vocabulary mismatch (`skill`/`plugin` vs `agent_skill`/`agent_plugin`) - resolved with an explicit mapping. Without this, no new ecosystem's entries - carry `resource_type` or distributions. -3. 15.9.1 harness resource scope and observation contract (shipped). -4. 15.10 acquisition plan/receipt/HMAC identity contracts (shipped surface). - -## Upstream contracts to pin before coding - -- DeepSeek Harness repository and releases: - -- dsh CLI entry point and install flow: `npx @deepseek-ai/dsh` (pin exact - package version; record install/plugin-management commands as documented at - the pinned release, not from memory) -- `dsh.plugin.json` manifest schema at the pinned release -- Plugin discoverability: the `dsh-plugin` GitHub topic - () and the semi-central hub list - (`dsh-external/hub`) — verify during implementation which of the two is - authoritative enough to crawl (decision gate below) -- Cordis kernel semantics (plugin identity, config entry tree, service keys): - and the Cordis paper - () - -Record exact tested versions and fixture provenance in the PR. dsh is a -developer preview: adapters must fail closed on an unsupported manifest or -state format/version rather than silently misattribute, and every dsh version -bump requires re-running the recorded-fixture suite before the pin widens. - -## Product contract - -### dsh as a harness - -Add `dsh` to the 15.9.1 harness adapter set (`codex`, `claude`, `hermes`, -`pi`, `opencode`, `custom`). The adapter must declare, with recorded fixtures: - -- native plugin locations and the Cordis declarative config entry tree that - registers a plugin (installing a plugin = the native manager adding an - entry; Logion never hand-edits the config tree in place of the manager); -- scope kinds supported (`repo-*`, `user`) and precedence, with the standard - default: repository root inside a Git worktree, no silent fallback to - `user`; -- observation capability: **none in this phase** — the adapter declares - `observation: unsupported` honestly; hooks are 15.11 scope; -- failure behavior for unsupported dsh versions, malformed manifests, and - missing state (fail closed, actionable error, no partial writes). - -### `dsh` distribution channel - -Extend the 15.10 channel set (`logion_bundle`, `npx_skills`, `npx_plugins`, -`hf`, `git`, `manual`) with: - -```text -dsh upstream Git-hosted Cordis plugin acquired through the dsh native flow -``` - -Channel rules inherited unchanged from 15.10: the channel is not the resource -identity; evidence attaches to `ResourceVersion.content_digest`; the same -plugin reachable as a Git repo and as a dsh plugin collapses into one resource -with multiple distributions; `native.argv` is a display/execution array, never -a shell string; the server never executes it. - -The acquisition plan's `native` block uses `tool: "dsh"`, the pinned tested -version, the documented install argv, the upstream locator, and the immutable -revision (40-char commit). `permissions` is populated from the manifest's -declared dependencies/services when the pinned schema exposes them, and is -labeled *declared by publisher, not verified* — this phase must not imply -enforcement or verification of declared capabilities. - -### Identity and reconciliation rules - -The 15.10 attribution ladder applies verbatim (exact Logion marker → native -manifest/lock plus immutable revision → content digest → canonical source plus -verified subpath → unresolved). dsh-specific sharpenings: - -- a Cordis plugin name or config entry `id` is **never** identity by itself; -- `dsh.plugin.json` is read only at a pinned `schema_version`-equivalent; - unknown versions quarantine the entry as `unsupported_manifest`, which can - never mint an `installation_id`; -- multiple candidate resources produce `ambiguous` with candidate IDs and no - attribution; name similarity is never identity; -- reconciling a plugin installed directly with dsh must not move, rewrite, or - reinstall it, and must not touch dsh's config tree or lock state. - -### The Logion dsh plugin (entry surface) - -A thin wrapper following the standing harness strategy -(`harness command → wrapper → logion CLI --json → render host-native -response`). Contract: - -- distributed the way real dsh plugins are distributed at the pinned release - (public Git repository tagged `dsh-plugin`, npm package if that is the - documented flow) so a dsh user discovers and installs it natively; -- declares the minimum Cordis dependencies it needs; it must not request - model, sandbox, or storage services it does not use; -- surfaces exactly: search/list, resource inspection with first-party evidence - and provenance, dry-run acquisition plan display, approval-gated acquire, - inventory, reconcile; -- contains no business logic: every operation shells to the public `logion` - CLI with `--json` and renders the result; no separate vendor code path; -- never proxies Logion credentials into the dsh context beyond invoking the - locally configured CLI, and never writes secrets into Cordis config entries; -- degrades honestly: when the `logion` CLI is absent it explains how to - install it and does nothing else. - -## Implementation - -### `logion` (public) - -- `packages/indexer/logion_indexer/adapters/dsh_hub.py` — discovery adapter - implementing the two-member adapter protocol, registered in - `cli.py:_get_adapter`, seeded in `seeds/sources.yaml` (the seed-name test - must resolve it). - **Decision gate:** implement against the cheapest authoritative source. If - `dsh-external/hub` (or the pinned-release equivalent) is an enumerable repo - list, crawl it as a plain repository target and defer GitHub topic search. - Add a `topic` mode to `github_direct` only if no enumerable list exists; - topic search uses the GitHub search API (different rate limits, mutable - ranked results) and must record discovery provenance per listing. -- Wire the dormant generic-resource path (`DiscoveredResource`, - `ResourceDedupPlan`, `_serialize_resource_item`) into `pipeline.py`/`cli.py` - and add the explicit indexer↔API `resource_type` mapping (dependency 2). -- `packages/cli/cli/_harness/dsh.py` — harness adapter per the 15.9.1 - contract. -- `packages/cli/cli/commands/resources/_channels/dsh.py` — channel adapter - (acquire via delegated native flow; discovery of installed state from dsh's - own manifest/config, mirroring the `skills_lock` fail-closed pattern). -- The Logion dsh plugin package/repository (location decided with the - distribution pin above; if in-monorepo, `packages/dsh-plugin/`). -- Reconcile support: `logion resources reconcile --from dsh` reading dsh - state without mutation. - -### `backend repository` - -- Add `CHANNEL_DSH` to `api/resources/constants/distribution_channels.py` and - the `CHANNEL_NATIVE_TOOL` map (`dsh` → `dsh`). -- Extend indexed upsert to create `dsh` (and `git`) distributions from - trustworthy source/package-map data with pinned revisions; ambiguous or - missing revision is quarantined or `manual`-only, as in 15.10. -- Reuse the 15.10 stable errors; `resource_native_tool_unsupported` and - `resource_native_tool_version_unsupported` must be reachable for unsupported - dsh versions (15.10 gap closure makes them exist). -- No new endpoints. No dsh-specific fields in the acquisition plan beyond the - existing generic `native`/`permissions` blocks. - -## Security and privacy - -- All 15.10 defenses apply: argv-only execution (no shell), sanitized - environment, cwd pinned, timeout, output cap, archive traversal defenses, - no secret logging. -- `dsh.plugin.json` and Cordis config content are untrusted input everywhere: - in the indexer, in reconciliation, and in the plugin wrapper. Manifest - fields never become CLI flags without validation. -- Cordis plugins are executable artifacts. This phase installs them only via - the native manager with explicit user approval; it never loads, executes, or - probes plugin code itself (execution/probing discipline arrives with the - 16.x evaluator/MCP-style lanes). -- Declared coeffects are displayed as publisher claims. A mismatch lane - (declared vs observed) is future evidence work; this phase must not label - declared capabilities as verified. -- The Logion dsh plugin sends nothing anywhere except through the local - `logion` CLI; no telemetry, no review submission, no usage claims (15.11). - -## Tests - -### Indexer - -- Recorded fixtures for the pinned hub/topic source: happy path, missing - manifest, unknown manifest version (quarantined with reason), duplicate - plugin across sources (single resource, multiple distributions). -- Generic resource path: dsh listings arrive with `resource_type` and a `dsh` - distribution row; vocabulary mapping round-trips; second run is - zero-change idempotent. - -### CLI - -- Harness adapter fixtures per 15.9.1: native locations, scope precedence, - repo/user isolation, unsupported-version failure, `observation: unsupported` - declared. -- Channel adapter: fake-executable tests asserting exact argv and no shell; - recorded real-`dsh` fixtures for install output/state at the pinned version; - fail-closed on unknown state format; pre-existing install reconciled without - mutation; ambiguous identity never links; second reconcile zero-change. -- Receipts: `channel: "dsh"` receipts carry `native_evidence` with - `manager_name: "dsh"`, pinned `manager_version`, canonical source, immutable - revision; digest/HMAC rules inherited from 15.10 unchanged. -- Install the same plugin in fixture repositories `xpto` and `acme`; distinct - receipts, no cross-repo or user-scope leakage. -- Launch a fresh real dsh session and assert native discovery of the exact - installed plugin (no fake dsh executable). - -### Logion dsh plugin - -- Loads in a real pinned dsh session; renders search/inspect/plan/inventory - from `logion ... --json` fixtures; absent-CLI degradation; no secret - material in any Cordis config entry it writes. - -## Mandatory proving-ground scenario - -Follow [the common real-agent gate](agent-proving-ground-phase-gate.md). Add a -builtin scenario named by behavior: - -**`builtin:dsh_plugin_discovery_install_and_reconcile`** — the scenario YAML, -fixtures, queries, and typed assertions land in -`logion/packages/agent-proving-ground/` with the implementation, per the gate -checklist. This phase is incomplete until the scenario passes with a real -cheap agent against the locally running API. - -- **Actors/fixtures:** one `operator` (developer persona). A dev-rig seed - target (`make proving-ground-seed SCENARIO=dsh_plugin_acquisition`) - publishes: an isolated HOME with the pinned real `dsh` CLI installed; the - Logion dsh plugin installable through dsh's native flow; fixture repository - `xpto`; two local Git fixture plugins with valid pinned-schema - `dsh.plugin.json` manifests, both indexed into the local catalog with `dsh` - distributions; and one poisoned fixture whose advertised digest mismatches - its content. No fake `logion`, `dsh`, or adapter executable. -- **Customer prompt (goal, not transcript):** "You use DeepSeek Harness. From - inside dsh, use the Logion plugin to find a lightweight capability for - repository xpto, check what Logion knows about it (source, version, digest, - license, declared permissions), install it into that repository — not - globally — and show what was installed. One plugin was already installed - here directly with dsh: make Logion recognize it too, without reinstalling - anything. Then prove a fresh dsh session actually loads what you installed. - Do not call Logion's HTTP API directly." -- **Flow:** discover through the Logion plugin → inspect → dry-run → approved - acquire via delegated native dsh flow → inventory → reconcile the - out-of-band install → rerun acquire (idempotent) → fresh dsh session - discovery. -- **Assertions:** `api.resource_acquisition_exists`, - `api.resource_distribution_selected` (allowed channel `dsh`), - `files.installed_artifact_digest_matches`, - `files.inventory_receipt_matches`, `api.native_install_reconciled` (the - direct dsh install attributed without reinstall), - `api.acquisition_idempotent`, plus a new typed assertion - `files.native_harness_discovers_installation` (fresh-session dsh state lists - the installed plugin with the inventoried digest); require `logs.no_500s`. -- **Negative case/evidence:** acquiring the poisoned fixture fails closed on - digest mismatch, creates no success receipt, and leaves dsh state untouched; - reconciliation of an unknown-manifest-version plugin yields - `unsupported_manifest`, never an attribution. Retain dsh version/output, - manifest digests, receipt/artifact digests, and proof of zero duplicate - state. -- **Drivers:** standard gate contract — `codex`/`gpt-5.4-mini` primary, - `claude-code`/`claude-haiku-4-5` fallback, `api_adapter: local-devrig`. - -## Acceptance criteria - -- [ ] An indexed dsh plugin discovered through Logion is acquired into a - repository scope through the real delegated `dsh` flow, with receipt, - digest, and HMAC identity per the 15.10 contract. -- [ ] A plugin installed directly with dsh is reconciled to the exact indexed - `ResourceVersion` without reinstalling, moving, or rewriting dsh state. -- [ ] The Logion dsh plugin installs through dsh's native flow from its public - distribution point and surfaces search → evidence → plan → acquire → - inventory → reconcile by shelling to the public CLI. -- [ ] A fresh dsh session discovers the exact installed plugins; installing in - `xpto` creates nothing in user scope or another repository. -- [ ] Unknown `dsh` versions and unknown manifest formats fail closed with - stable errors and can never mint an `installation_id`. -- [ ] Every dry-run remains zero-write and shows channel, revision, digest, - declared permissions (labeled as publisher-declared), and exact argv. -- [ ] All evidence shown in the dsh surface is labeled first-party; no surface - claims or implies independent/network verification. -- [ ] Existing harness adapters, channels, and `skills`/`resources` CLI - behavior remain compatible. -- [ ] The mandatory dogfood artifact exists with one real Logion-mediated dsh - acquisition and one real out-of-band reconciliation. - -## Rollout - -1. Indexer discovery + catalog listings with `dsh` distributions (read-only - reach; no acquisition). -2. Reconciliation of existing dsh installs (`--from dsh`) without executing - the manager. -3. Delegated `dsh` acquisition behind a per-channel feature flag. -4. Logion dsh plugin published through the native distribution point. - -Metrics mirror 15.10 (planned/started/succeeded/failed acquisition, -verification level, reconcile matched/ambiguous/drifted, native tool/version, -channel) — no user project paths. Each dsh version bump re-runs recorded -fixtures before the tested-version pin moves. - -## Out of scope - -Usage observation, hooks, feedback, and reviews for dsh (15.11/15.11.1); -signed portable evidence (15.13); evaluation of dsh plugins, including -comparative service-key evaluation (16.2+); MCP-style safe probes of plugin -code (16.11 pattern); any DeepSeek-specific protocol contract, catalog schema, -or receipt format (17.1 upstream-proposal territory); mirroring or hosting dsh -plugin artifacts; executing or sandbox-probing plugin code. diff --git a/plans/phase-15.11-native-use-observation-linked-feedback-and-reviews.md b/plans/phase-15.11-native-use-observation-linked-feedback-and-reviews.md deleted file mode 100644 index de2ba023..00000000 --- a/plans/phase-15.11-native-use-observation-linked-feedback-and-reviews.md +++ /dev/null @@ -1,648 +0,0 @@ - - -# Phase 15.11 — Native-use observation, linked feedback, and reviews - -> **Implementation status (2026-08-22): built, partially proven.** The usage -> spool, receipt and feedback APIs, integrations commands, harness observation -> hooks (Claude Code, Codex), Hermes lifecycle observation adapter, -> Hermes/Pi explicit-report fallback, consent enforcement, and -> consent-driven upload are implemented. The phase is **not complete**: -> the proving-ground evidence still relies on replaying a recorded -> `PostToolUse` payload into the installed hook rather than the harness -> delivering that payload live, and the remaining "Still open" items below -> are not yet closed. The existing CLI inventory scan is not proof of use and -> must not be described as observation telemetry. -> -> **Gate:** both mandatory scenarios passed with a real agent driver on -> 2026-08-17 (`claude-code`/`claude-haiku-4-5`, local-devrig) against the -> tightened scenario and projection assertion; the cross-repo audit reports -> no critical findings. The observation in that evidence comes from piping -> the harness's recorded `PostToolUse` payload into the installed hook, not -> from that hook firing live — a live hook needs a `logion` on PATH that -> carries `usage observe`. -> -> **Still open before this phase may be called complete:** -> 1. Prove a live hook end to end, with the harness itself delivering the -> payload to an installed CLI, rather than the documented replay -> fallback. The current phase-gate evidence still uses replay because -> cross-driver delivery is not yet supported by any harness. -> 2. Pseudonymous participation: server-side `shadow` identity tier support -> exists, but the contract was tightened to require a signer-capable local -> pseudonymous subject; the public CLI still has no local keypair-backed -> pseudonymous identity flow. -> 3. Published first-party artifacts: `packages/harness-plugins/` now -> contains the observer scaffold, but no official `npx skills add` -> companion or `npx plugins add` observer package is published on the -> stable coordinates, so the clean-machine onboarding contract remains -> unbuilt. -> -> **Dogfood — Level 2 (real use and feedback):** the implementing agent acquires a resource through any supported channel, uses it in its ordinary harness, and submits feedback through Logion linked to the exact original `ResourceVersion`. -> **After this phase:** Logion can learn from resources installed by `npx skills`, `npx plugins`, `hf`, or Logion itself without forcing a new acquisition workflow. -> **Honesty boundary:** observation means “probably used”; feedback means “an agent/user reported an outcome”; neither is a controlled evaluation or universal quality claim. - -## Product thesis - -This is the first compounding data loop: - -```text -native install → exact inventory attribution → harness observes use -→ local pending usage → agent/user reports outcome -→ generic resource feedback → optional Course-review projection -→ aggregate demand/failure signal → funded improvement -``` - -The highest-value early signal is not “which resource is popular in a catalog?” It is “which exact versions people actually use, for which task classes, with what outcome and friction?” - -## Mandatory dogfood prompt for the implementing agent - -Run after the phase is implemented: - -```text -You are implementing Phase 15.11. Prove that feedback works when Logion did not -perform the installation. - -1. Search Logion for a resource relevant to this phase: - `logion listings search --query "agent hooks privacy telemetry feedback" - --include-indexed --limit 5 --json`. -2. Select an external Agent Skill with an exact `npx_skills` distribution and a - published Course/Resource with a Logion bundle. -3. Ask for approval. Install the external skill directly with the exact displayed - `npx skills add ...` command, not `logion resources acquire`. -4. Run `logion resources reconcile --from skills --scope repo-root --json`; require an - exact resource/version match. -5. Enable Logion observation for the current harness with - `logion integrations enable HARNESS --dry-run`, inspect the config diff, ask for - approval, then enable it. -6. Use the externally installed skill on a bounded Phase 15.11 implementation task. -7. Run `logion usage pending --json`; verify the exact resource/version/channel. -8. Submit: - `logion feedback submit RESOURCE_ID VERSION_ID --rating N --usefulness N - --reliability N --tool-safety N --token-efficiency N - --completed-task|--not-completed-task --task-class software-development - --body "Resource-focused feedback with no repository-private data" --json`. -9. If that resource/version projects to a CourseVersion and review eligibility allows - it, verify the response contains `course_review_projection`. Otherwise verify that - generic feedback succeeded without fabricating a marketplace buyer review. -10. Repeat one use with observation disabled and prove zero hook spool/write/upload. -11. Save `artifacts/dogfood/phase-15.11.md` with acquisition channel, inventory record, - hook diff, observation IDs, pending output, consent mode, feedback ID, projection - disposition, and privacy canary result. -``` - -The phase fails if dogfood requires reinstalling the external skill through Logion. - -## Dependencies - -- The 15.9.1 harness resource scope and observation requirements carried - forward as a contract; they are not a claim that observation shipped in - 15.9.1. -- 15.9 resource/version/projection identity. -- 15.10 local acquisition inventory and reconciliation. -- Existing CourseReview API, auto-review consent, pseudonymous/agent identity, CLI local state/redaction, and supported harness projections. - -## Upstream contracts to pin before coding - -- `skills` install/use/list/update semantics and lock/state format: -- `skills` telemetry opt-out contract: -- Vercel's plugin installation/observer precedent: and -- Hugging Face CLI and agent-skill entry surface: -- Codex scopes: -- Claude Code scopes: -- Hermes skills/profiles: and -- Pi skills: - -Logion's consent must be independent from upstream telemetry preferences but respect the strongest local opt-out signal where reasonable (`DO_NOT_TRACK`, upstream disable flag, and Logion's own explicit `off`). Never assume upstream consent implies Logion consent. - -## Separate the three signal classes - -### 1. Local observation - -A local, privacy-minimized hint that a harness referenced an installed resource. It is not uploaded unless consent policy permits. - -### 2. Usage receipt/telemetry - -A structured, rating-free statement that an attributed resource participated in a task/session. It may include coarse task class, harness, outcome known/unknown, and counters/buckets. It is opt-in, pseudonymous where allowed, and never becomes a review automatically. - -### 3. Feedback/review - -An intentional post-task report containing subjective rating dimensions and completed-task status. It may be agent-authored under the user's configured policy. It becomes a Course review only through explicit projection rules. - -Do not collapse these tables or labels. - -## Local observation architecture - -```text -native manager / Logion inventory - │ exact installed paths + resource/version - ▼ -harness hook/plugin ── minimal event ──▶ logion usage observe - │ - ▼ - resolve against local inventory - │ - ▼ - $LOGION_HOME/usage/observations.jsonl - │ - pending/session-end feedback prompt -``` - -Raw hook payloads, prompts, commands, file contents, paths, and tool arguments exist only in memory long enough to resolve attribution. They are not written to the spool or sent. - -Every native integration writes the same versioned local envelope to the shared -Logion spool. Harness plugins and hooks never call the remote API directly: the -CLI owns attribution, consent, redaction, retry, and upload. - -## Observation schema - -`logion/packages/cli/cli/usage/observations.py`: - -```python -@dataclass(frozen=True) -class UsageObservation: - schema_version: Literal[1] - observation_id: str - observed_at: str - harness: str - event: Literal["resource_invoked", "resource_file_read", "resource_tool_used"] - resource_id: str - version_id: str - resource_type: str - acquisition_channel: str - installation_id: str - scope_kind: Literal[ - "repo-current", "repo-parent", "repo-root", - "user", "admin", "system", "custom" - ] - scope_id: str - session_hash: str | None -``` - -No free-text or path field is allowed. A test pins the dataclass/schema fields. -`installation_id` and `scope_id` are the opaque, profile/node-scoped HMAC -identifiers defined by 15.9.1. Observation code consumes the IDs stored in the -validated local acquisition receipt; it never recomputes them with a plain path -hash and never serializes the underlying local root/path. - -Deduplicate `(session_hash, resource_id, version_id, event)` within a bounded window. Unknown/ambiguous local attribution is dropped locally with an optional debug counter; never guess. - -## Integration surfaces - -### First-party observer packages - -Ship the same observer through native workflows: - -- a Logion companion Agent Skill installable through `npx skills add --skill logion`; -- a Logion observer plugin installable through `npx plugins add `; -- the existing Logion CLI/companion installer; -- documented thin hooks for harnesses not supported by the plugin manager. - -The official source coordinates must be filled with the actual published repository/package before implementation; placeholder coordinates cannot ship. - -The skill teaches the agent to inspect `logion usage pending`, submit feedback after meaningful use, and respect user consent. The plugin/hook observes lifecycle/tool events. Neither bundles API secrets. - -### Clean-machine onboarding contract - -`npx skills add` installs skill files, not a trusted system binary. Do not conceal a binary installation inside skill activation. - -On a machine with a supported agent and Node but no Logion CLI: - -1. `npx skills add --skill logion` installs the signed/pinned Logion companion skill through the user's existing manager. -2. On first relevant activation, the skill performs a read-only `command -v logion`/version check. -3. If missing, it explains why the local CLI is required for inventory, privacy filtering, hooks, feedback, and credentials; it displays the official installer/version/checksum path and asks for approval. -4. After approval, it uses the existing official Logion installer and verifies version/checksum. It must not execute an unpinned arbitrary `curl | sh` assembled from marketplace metadata. -5. It runs `logion integrations detect`, shows the exact enablement/config diff and consent modes, then asks before enabling. -6. The user may keep only the companion skill without telemetry/feedback; no hook or upload consent is implicit. - -On a machine with `npx plugins`, the official Logion plugin may bundle the thin observer/hook code supported by the plugin format, but still delegates identity, inventory resolution, consent, spool, redaction, and API writes to the verified Logion CLI. If the CLI is absent it follows the same explicit bootstrap contract. - -Target one-command entry surfaces: - -```bash -npx skills add OFFICIAL_LOGION_SOURCE --skill logion -npx plugins add OFFICIAL_LOGION_PLUGIN -``` - -These commands must appear on the landing/README with truthful qualifiers: the first installs the companion skill; the second installs the observer plugin; neither silently opts the user into upload or auto-feedback. - -### Published artifact requirements - -- Official source/release ownership and stable coordinates. -- Immutable release tag/commit and content digest recorded as the Logion `ResourceVersion`. -- Package map identifies companion, bootstrap references, supported harnesses, and required `logion` CLI range. -- Provenance/SBOM/checksums for plugin/npm artifacts. -- Installation fixtures for project/global, supported agents, update, uninstall, and no-CLI bootstrap. -- `npx skills update` / plugin updates preserve user consent and Logion inventory attribution. - -### CLI commands - -```bash -logion integrations detect [--json] -logion integrations enable HARNESS [--dry-run] [--mode prompt|auto|local-only] -logion integrations disable HARNESS -logion integrations status [--json] -logion usage observe --harness HARNESS --stdin -logion usage pending [--since 24h] [--json] -logion usage dismiss OBSERVATION_GROUP_ID -logion feedback submit RESOURCE_ID VERSION_ID ... -logion feedback list --mine [--json] -``` - -`usage observe`: - -- exits 0 on parse/resolution/spool failure so it never breaks the harness; -- reads at most 1 MiB and finishes within two seconds; -- writes fixed-schema records only; -- with feedback/telemetry off, performs zero read/write/network work after config check; -- logs diagnostics only under explicit debug. - -### Supported harness strategy - -- Prefer the `npx plugins` format when the current agent supports it. -- For Claude Code/Codex/other hook-capable harnesses, marker-keyed merges install one thin hook that writes the shared local observation envelope. -- For unsupported harnesses, the companion may record explicit resource activation/use from the agent's own workflow; it must not scrape shell history. -- Configuration edits are `--dry-run`, idempotent, preserve unknown user config, and uninstall only `_logion_managed` entries. - -The implementer must verify each harness's current hook schema against official documentation and commit recorded fixtures. Never trust old field names from this plan. - -The release matrix is explicit: - -| Harness | Repository resources | User resources | Observation path | -|---|---|---|---| -| Codex | `.agents/skills` from CWD through repository root | `~/.agents/skills` | supported extension/hook where available; otherwise explicit companion report | -| Claude Code | `.claude/skills` | `~/.claude/skills` | native plugin/hooks | -| Hermes | repository-specific `skills.external_dirs`/profile | `~/.hermes/skills` | lifecycle integration or explicit companion report | -| Pi | `.pi/skills` or `.agents/skills` from CWD through root | `~/.pi/agent/skills` or `~/.agents/skills` | extension where supported; otherwise explicit companion report | - -An adapter is “supported” only after a recorded real-harness fixture and its -customer-like proving scenario pass. Directory scanning alone is inventory, -not use observation. - -### Pre-existing remote and closed MCP resources - -The user may install a vendor plugin or remote MCP connector through its -original marketplace/client and install the Logion companion separately. -Logion must then adapt to that native installation: - -- inventory the harness's plugin-manager state and MCP configuration without - rewriting either; -- reconcile the original publisher, public plugin/source revision, declared - remote endpoint, manifest, and available digest to the exact - `ResourceVersion`; -- preserve the vendor plugin, MCP server, and each local installation as - distinct distribution/resource/installation identities; -- attribute supported harness tool-use events to the original publisher's - resource rather than creating a Logion-owned wrapper or duplicate listing; -- exclude OAuth tokens, prompts, tool arguments/results, documents, local - paths, and arbitrary model context from inventory and observations. - -The default observation path is a consented Logion hook/plugin beside the -vendor integration. Logion does not silently proxy TLS, replace the MCP -endpoint, inject itself into OAuth, or require reinstalling the vendor -resource. After the user enables the Logion integration and selects a consent -mode, supported hooks may reconcile and spool minimal events automatically. -If a harness exposes no trustworthy local tool-use event, Logion reports -inventory-only support and may offer explicit prompt-mode feedback; it must -not infer invocation from installation, connector availability, quota changes, -or network traffic. - -Normal authorized use may produce privacy-minimized observations and -intentional feedback. Active Logion-run probes/evals against a remote server -remain governed by the owner opt-in, public-test policy, terms, credentials, -and synthetic-input controls in Phase 16.11. User authorization to use a -service is not permission for Logion to benchmark or probe it independently. - -## Feedback API contract - -### Database - -Add: - -```text -resource_usage_receipts - id UUID PK - resource_id, resource_version_id - reporter_agent_id / pseudonymous_subject_id - identity_tier shadow|account|verified - acquisition_channel - task_class VARCHAR(64) - harness VARCHAR(64) - outcome completed|not_completed|unknown - coarse counters/buckets JSONB - consent_policy_digest - observed_at, submitted_at - receipt_digest UNIQUE per reporter - -resource_feedback - id UUID PK - resource_id, resource_version_id - reporter_agent_id / pseudonymous_subject_id - identity_tier - acquisition_channel - rating SMALLINT - usefulness, reliability, tool_safety, token_efficiency NUMERIC - completed_task BOOLEAN - task_class VARCHAR(64) - body TEXT NULL - created_at, updated_at - source_receipt_id NULL - UNIQUE(reporter_subject, resource_version_id, task_class) - -resource_feedback_course_projections - feedback_id UUID PK/FK - course_review_id UUID NULL - disposition projected|not_a_course|ineligible|self_review|paid_entitlement_missing - created_at -``` - -Keep CourseReview as the marketplace-facing commercial review. Generic feedback is the source for all resource types and acquisition channels. - -### API - -- `POST /v1/resources/{resource_id}/versions/{version_id}/usage-receipts` -- `POST /v1/resources/{resource_id}/versions/{version_id}/feedback` -- `GET /v1/resources/{resource_id}/feedback` -- `GET /v1/resources/{resource_id}/feedback/summary` -- `GET /v1/feedback/mine` - -Use `api/resource_feedback/{repositories,services,controllers,responses}/`. Projection calls the existing Course review domain service; never writes CourseReview directly. - -Stable dispositions/errors: - -```text -feedback_resource_version_mismatch -feedback_acquisition_not_attributed -feedback_invalid_score -feedback_body_private_data_detected -feedback_self_review_blocked -feedback_course_projection_ineligible -feedback_paid_entitlement_required -usage_receipt_consent_required -usage_receipt_duplicate -``` - -## Projection rules - -Generic feedback projects to CourseReview only when all are true: - -- exact `ResourceVersion → CourseVersion` projection exists; -- reporter is not the owner/author under existing self-review rules; -- Course/version is reviewable; -- free/open Course policy or active paid entitlement permits a marketplace review; -- feedback is intentional, not synthesized solely from passive observation; -- reporter has not already projected an equivalent feedback row. - -External installation of an open resource can produce generic feedback. It does not create a paid entitlement. If the same artifact is a paid Course, the generic feedback remains visible in the resource evidence surface but is not mislabeled “verified buyer review.” - -The API response explains the projection disposition. - -## Identity and consent - -### Pseudonymous participation - -- Free resource discovery, reconciliation, local observation, and local pending usage do not require signup. -- A random local pseudonymous subject may sign/upload feedback only after explicit consent policy is stored. **Sign is literal.** The subject is a locally held keypair with a stable public identifier, not an opaque random id — an id satisfies "works without an account" and forecloses everything downstream, because an agent that cannot sign can never become an issuer without an identity migration. This is the cheapest decision in the phase and the most expensive one to reverse. -- If the backend supports provisional agents, the local subject attaches later to an account without changing historical IDs. -- Identity tier is carried into aggregation; it is not a hidden multiplier. -- Money, authorship, claims, and paid entitlement stay account-gated. - -### Consent modes - -```text -off no observation spool, upload, feedback prompt, or network call -local-only local observation/pending; no upload -prompt local observation; agent asks before each feedback submission -auto user explicitly opted in; agent may submit one post-task feedback report -``` - -`auto` does not infer a rating from file reads. The agent must have task outcome context and must generate the structured report at meaningful task completion. Users can inspect, edit, delete, export, or disable. - -An unconfigured harness is `off`, not `local-only`: `effective_mode` returns -`off` for anything never configured, and `DO_NOT_TRACK` forces it regardless. -Observation is opt-in at the harness, not merely at the upload. - -### One knob is the wrong shape for two questions - -The two signal classes are separated everywhere in this design — different -tables, different endpoints, `Rating? Never` against `Rating? Yes` — and then -joined again by a single per-harness `mode`. `may_spool` and `may_upload` both -derive from it, so consenting to a receipt and consenting to an agent-authored -review are the same act. - -They are not the same ask: - -| | Deterministic receipt | Agent review | -| --- | --- | --- | -| Cost | zero tokens; a hook is a subprocess | a model call, once per task | -| Content | version ran, completed or failed, duration bucket, harness | rating, four subjective dimensions, prose | -| Subject | the artifact | work done in the user's repository | - -"May I record that this version ran and completed" is a small request most -people grant. "May my agent write a public review of what happened in my -repo" is a large one. Binding them to one switch loses the easy consent to -pay for the hard one, and that is the most likely single cause of the empty -cohort that Loop B has to answer around on day one. - -Split the mode into a receipt scope and a review scope. Neither implies the -other, and the review scope stays account-gated because it carries prose. - -**Why the default does not become `auto`.** More rows would be worse data, -not better. `next-steps.md` requires publishing consent mode alongside `n`, -so every row carries its provenance permanently, and a cohort marked -default-on is discounted by exactly the careful reader the project is for. -16.8 suppresses below minimum cohort and caps concentration anyway, so volume -from a homogeneous default-on population frequently cannot be published at -all — the reputational cost paid for a number that stays suppressed. `auto` -also fires without the user confirming an outcome, which biases toward -completion. A measurement whose provenance is "they did not turn it off" is -not evidence, which is the category this project exists to improve on. - -The cold start is real, and the answer is when the question gets asked, not -whether. After the first meaningful use, show the exact pending receipt and -ask once. That converts better than a silent default and the row comes out -marked as explicit consent, which is publishable. - -## Agent companion behavior - -At resource/task completion: - -1. Read `logion usage pending --json`. -2. Match only resources actually used in the current session. -3. If `prompt`, present resource/version, channel, task class, proposed scores/body, and ask. -4. If `auto`, submit once under stored consent. -5. If task outcome is unknown or resource only read incidentally, leave pending or dismiss—do not rate. -6. Record the feedback ID/tombstone locally so repeated hooks do not create repeated reviews. - -Review body contains resource-focused observations only, never prompt, repository name, code, customer data, or personal information. - -## Aggregation read model - -Resource feedback summary exposes: - -- unique reporter count by identity tier; -- attributed use/feedback count by acquisition channel; -- completed-task rate with sample size; -- rating dimensions with sample size; -- task-class distribution under minimum-cohort privacy threshold; -- recent version coverage and confidence/limitations; -- Course-review projection counts separately. - -Do not rank yet. Do not call passive receipt volume “users” without explaining pseudonymous subjects and dedup. - -## Files to change - -### Public repository - -- `packages/cli/cli/_local_state.py` or split inventory/usage modules. -- New `packages/cli/cli/usage/`. -- New `packages/cli/cli/commands/integrations/`. -- New `packages/cli/cli/commands/feedback/`. -- Update `commands/courses/report_usage.py` to write the local reported tombstone after API success. -- Update companion `SKILL.md` usage flow; keep context addition compact. -- Add `packages/harness-plugins/` or official plugin-format package. -- Publish official Logion skill/plugin acquisition artifacts and provenance. -- Client Resource feedback APIs/types. - -### Private repository - -- Migration/models for receipts, feedback, projections, pseudonymous subject if not already supported. -- `api/resource_feedback/`. -- Reuse Course review validation/projection service. -- Resource feedback summary/search integration. -- Rate limits, abuse detection, deletion/export, observability. - -## Required tests - -- Exact path/source/digest inventory resolution; ambiguous basename/source drops. -- Hook parsers with recorded official payload fixtures for every supported harness. -- Two repositories with the same resource produce distinct installation/scope - IDs; observation in one cannot attach to the other or to user-global state. -- Real Codex, Claude, Hermes, and Pi sessions cover native discovery plus the - best supported observation path; unsupported event types fail locally. -- Plugin-format install/uninstall/status and marker-keyed config merges. -- `off` means byte-identical local state and zero network; `local-only` means zero upload. -- Raw prompt, command, path, repo, env, secret, and tool payload canaries never reach spool/request/log. -- Concurrent append, torn line, rotation, dedup, unknown schema, clock skew. -- Pending grouping/tombstone/dismiss and one-shot semantics. -- Generic feedback for Logion bundle, direct `npx skills`, `npx plugins`, and `hf` acquisition records. -- Pre-existing vendor plugin plus OAuth remote MCP reconciliation: exact - publisher/resource/version attribution when evidence supports it, no - reinstall/reconfiguration/proxy, and typed inventory-only behavior when the - harness exposes no trustworthy tool-use hook. -- Course projection happy/free, no Course, paid entitlement missing, self-review, duplicate upsert. -- Pseudonymous/account attach and identity-tier aggregation. -- Sybil/rate-limit and minimum-cohort privacy fixtures. -- End-to-end direct native install → reconcile → observe → pending → feedback → projection disposition. - -## Rollout - -1. Local-only observation on Logion's own development harnesses. -2. Prompt-mode generic feedback from the team. -3. Published Logion companion skill and observer plugin. -4. Invite-only external prompt-mode beta. -5. Pseudonymous upload after privacy/abuse review. -6. Auto mode only for users who explicitly enable it. - -Metrics: integration enabled/disabled, attribution exact/ambiguous/dropped, pending/feedback conversion, channel, identity tier, projection disposition, failure code, deletion/opt-out. Never include paths/prompts/task contents. - -## Mandatory proving-ground scenario - -Follow [the common real-agent gate](agent-proving-ground-phase-gate.md). Add -`builtin:native_use_observation_and_feedback`. - -- **Actors/fixture:** two fresh agent processes have isolated homes and work in - repositories `xpto` and `acme`; an `operator` observes API state. The seed creates a real - `npx skills`-compatible review helper, and the first session installs the - official Logion integration with its documented customer command. -- **First prompt:** “Install the indexed review helper for this repository only - and use it to review repository xpto. Keep telemetry at the default privacy - mode. Do not install it globally.” -- **Second prompt:** “Continue my work. Show feedback Logion queued from - capabilities I actually used, let me inspect exactly what would be sent, and - submit an honest review linked to the original resource.” -- **Assertions to implement:** `files.native_use_observed`, - `files.feedback_pending`, `api.resource_feedback_exists`, - `api.feedback_linked_to_acquisition`, - `api.course_review_projection_exists`, - `api.raw_observation_not_uploaded`, and - `api.feedback_submission_idempotent`; also assert no observation from `xpto` - attaches to the same resource installed in `acme`. -- **Consent/evidence:** session one uploads no prompts, repository content, or - raw events. Retain integration version, observation/pending receipt IDs, - consent mode, feedback ID, source link, redacted payload, and no-500 proof. - -Add `builtin:remote_private_mcp_feedback`: - -- **Fixture:** a public vendor plugin manifest points to an OAuth-protected - remote MCP fixture whose implementation and data are unavailable to Logion. - The vendor connector is installed first through the native manager; the - Logion companion is installed and enabled separately. -- **Prompt:** “Use the already-installed vendor connector for this task. Let - Logion attribute its use to the original vendor resource without reinstalling - it, changing its endpoint, or recording the request or response.” -- **Assertions to implement:** `files.remote_mcp_reconciled`, - `files.vendor_install_unchanged`, `files.no_mcp_proxy_installed`, - `api.remote_mcp_use_attributed`, `api.original_publisher_preserved`, - `api.remote_mcp_feedback_linked`, and - `api.remote_mcp_private_payload_not_recorded`. Run the same fixture in a - harness without tool-use hooks and require an explicit - `inventory_only_observation_unsupported` result with no fabricated event. -- **Evidence:** retain public manifest/source/endpoint/version digests, local - installation and observation IDs, consent mode, redacted feedback ID, and - before/after hashes proving that vendor configuration was not rewritten. - -## Acceptance criteria - -Every criterion names the check that proves it. `assertion:` is a proving-ground -assertion this phase's gate requires, `test:` is a test path in either checkout, -and `deferred:`/`unspecified:` are declarations of debt that the generated -`DEFERRED.md` collects. The auditor rejects a criterion with no -marker and a marker naming an assertion the gate does not require. - -- [ ] A skill installed directly by `npx skills add` can be observed and receive feedback linked to the exact Logion `ResourceVersion` without reinstalling through Logion. - (proof: assertion:api.feedback_linked_to_acquisition) -- [ ] A Logion-hosted Course, Vercel plugin, and HF revision use the same generic feedback contract. - (proof: test:packages/api/tests/resource_feedback/test_feedback_projection.py::test_projection_to_course_review_for_eligible) -- [ ] A pre-existing vendor plugin backed by a closed OAuth remote MCP server - can be reconciled and, where the harness exposes a trustworthy hook, - observed through a separately installed Logion companion without proxying, - reinstalling, or claiming ownership of the original resource. - (proof: assertion:api.remote_mcp_use_attributed) -- [ ] Passive observation never creates a rating or CourseReview. - (proof: test:packages/api/tests/resource_feedback/test_usage_receipt.py::test_usage_receipt_never_projects_feedback_or_course_reviews) -- [ ] Eligible feedback projects through the existing Course review service; ineligible feedback remains useful and clearly labeled. - (proof: assertion:api.course_review_projection_exists) -- [ ] `off` produces no local observation state and no network request. - (proof: test:packages/cli/tests/test_usage_upload.py::test_off_never_uploads) -- [ ] The official Logion skill/plugin integrates feedback into a user's existing agent workflow. - (proof: deferred:no official companion skill or observer plugin is published) -- [ ] Codex, Claude, Hermes, and Pi have tested native scope discovery and a - declared observation path or an honest explicit-report fallback. - (proof: test:packages/cli/tests/test_integrations_commands.py) -- [ ] From a clean machine, `npx skills add OFFICIAL_LOGION_SOURCE --skill logion` installs the companion; first use can install/verify the CLI with explicit approval and then preview/enable a harness integration. - (proof: deferred:the companion scaffold is merged but the official source coordinates are still unpublished, so the clean-machine onboarding contract is unbuilt) -- [ ] `npx plugins add OFFICIAL_LOGION_PLUGIN` installs the observer integration without duplicating identity, inventory, spool, redaction, or API-write logic. - (proof: deferred:the observer scaffold is merged but no official observer plugin package is published) -- [ ] Neither native install command silently opts the user into observation upload or automatic feedback. - (proof: deferred:blocked on the two unpublished native install commands above) -- [ ] Dogfood submits a real feedback record for an externally installed resource and records the Course projection disposition. - (proof: test:artifacts/dogfood/phase-15.11.md) -- [ ] The observation hook fires from the harness itself, not from a replayed - `PostToolUse` payload. - (proof: deferred:cross-driver payload delivery is unsupported, so the recorded evidence used the documented replay fallback) -- [ ] Free discovery and feedback work without an account, through a locally - held **keypair** whose public identifier stays stable when it later - attaches to an account. A random opaque id does not satisfy this: the - subject has to be able to sign. - (proof: test:packages/api/tests/resource_feedback/test_submit_feedback.py::test_feedback_controller_accepts_signed_pseudonymous_subject) -- [ ] Generic self-review is detectable for every resource type, not only for - resources that project to a `CourseVersion`. - (proof: test:packages/api/tests/resource_feedback/test_submit_feedback.py::test_submit_feedback_self_review_still_records) -- [ ] Exactly one observation envelope is normative for a usage record. - (proof: test:packages/cli/tests/test_observation.py) -- [ ] Consenting to a deterministic receipt and consenting to an - agent-authored review are separate scopes; neither implies the other. - (proof: test:packages/cli/tests/test_integrations_commands.py::test_status_keeps_receipt_and_review_scopes_separate) - -## Out of scope - -ARD discovery (15.12), signed scanner evidence (15.13), sponsorship selection (15.14), isolated execution (15.15), controlled evals, universal reputation, covert telemetry, or automatic funding. diff --git a/plans/phase-15.11.1-publisher-integrated-consented-observation.md b/plans/phase-15.11.1-publisher-integrated-consented-observation.md deleted file mode 100644 index f760c883..00000000 --- a/plans/phase-15.11.1-publisher-integrated-consented-observation.md +++ /dev/null @@ -1,734 +0,0 @@ - - -# Phase 15.11.1 — Publisher-integrated consented observation projections - -> **Implementation status (2026-08-24): not built.** No `logion instrument` -> command, no instrumentation-profile schema, no reporter, no publisher-receipt -> ingestion. The parts that exist were built for this phase from 15.11: the -> `shadow` identity tier, the `outcome`/`task_class`/`duration_bucket`/ -> `integration_version` fields on the observation envelope, and -> `consent_policy_digest` on usage receipts. -> -> **This revision corrects the phase's premise.** The previous version assumed -> Logion could generate native projections carrying a bundled reporter that -> runs automatically wherever the resource is installed. That is not what the -> distribution formats permit, and the plan contradicted itself: its own -> capability table already said metadata in `SKILL.md` is a declaration, not an -> execution. The mechanism below is narrower and buildable; the honesty -> boundary is now the same in every section. -> -> **Dogfood — Level 2.1 (publisher-side adoption):** a resource owner adds one -> Logion instrumentation profile, publishes a native projection, and a fresh -> user accepts one precise disclosure during install or first activation. -> Subsequent supported uses emit only the approved minimum receipt without -> requiring the full Logion CLI or a second Logion installation. -> **After this phase:** Logion can enter through the resource publisher's -> existing distribution instead of convincing every end user to adopt the whole -> Logion suite. -> **Honesty boundary:** an install is not use, activation is not success, and a -> publisher-integrated receipt is first-party field evidence rather than an -> independent evaluation. - -## Why this subphase exists - -Phase 15.11 makes a separately installed Logion companion observe resources -already present in a harness. That remains the universal opt-in path, but it -leaves a difficult adoption step for an individual who only wants one skill, -plugin, MCP connector, or model. - -This subphase adds the complementary publisher path: - -```text -publisher owns resource/version - → Logion generates a thin native projection - → user installs the resource through its normal manager - → manager or first activation shows one exact telemetry disclosure - → user accepts, chooses local-only, or declines - → the bundled reporter emits approved minimum receipts - → Logion reconciles them to the original ResourceVersion - → evidence, feedback, scorecards, and improvement candidates accumulate -``` - -The resource remains useful when Logion is unavailable or consent is denied. -The full CLI remains valuable for operators, evaluators, runners, contributors, -and publishers, but it is not a per-user prerequisite for this narrow receipt -path. - -## The premise correction - -Three upstream facts, verified 2026-08-24, bound what this phase may claim. - -1. Agent Plugins 1.0.0 is a **directory format**, not an npm package: a - `plugin.json` manifest, `skills/` holding Agent Skills, an optional - `mcp.json`, and reverse-domain **client-specific namespaces** that carry - hooks. `npx plugins add owner/repo` installs it; the npm package `plugins` - is the installer, not the distribution channel. - (, - ) -2. The portable core is therefore **declarative**. Nothing in it guarantees - that any client executes code. Execution lives in the client namespaces, - which means it is a per-client capability, pinned per client release. -3. A static Agent Skill is markdown. Metadata a manager tolerates is not - metadata a harness executes. - -The consequence is that "the publisher ships a reporter and telemetry flows" -is only true for a client whose hook contract Logion has pinned against an -exact release with a recorded payload fixture. Everywhere else it is false, and -a plan that promises it everywhere produces an implementation that infers use -from installation. - -So the phase declares three tiers, and the tier decides what the projection may -claim. A projection never upgrades itself at runtime. - -| Tier | Condition | May claim | -| --- | --- | --- | -| `hook` | the client's hook contract is pinned to an exact release, a payload fixture is recorded in the repository, and the reporter runtime is present | automatic use observation | -| `explicit_report` | no lifecycle hook, but the skill's own text can instruct the agent to report use through a documented command | reported use only | -| `unsupported` | no pinned hook, or no reporter runtime, or the pinned release drifted | nothing — inventory and install facts only | - -A tier is resolved per client, so the phase declares its client coverage -explicitly rather than promising "supported harnesses". - -| Client | How Logion delivers the observer | Tier | Events it may honestly report | Payload fixture | Gate | -| --- | --- | --- | --- | --- | --- | -| Claude Code | client-namespace hook entry → `settings.json` `PostToolUse` | `hook` | activation and terminal outcome | recorded (`claude_code_post_tool_use.json`) | required | -| Codex | client-namespace hook entry → `.codex/hooks.json` `PostToolUse` | `hook` | activation and terminal outcome | recorded (`codex_post_tool_use.json`) | required | -| Hermes | generated stdlib-only Python plugin under `~/.hermes/plugins/logion-observer/`, registered through the harness's `register_hook`, enabled in `~/.hermes/config.yaml` | `hook`, **user scope only** | `resource.activated` **only** | **must be recorded — none exists yet** | **required** | -| DeepSeek Harness | `@logionsh/dsh-plugin`, a Cordis bundle whose patch loads the package by name | `explicit_report` today | use reported through the `logion_*` tools | not recorded | not gate-required | -| Everything else | portable static skill, no observer | `unsupported` | none | — | — | - -Three of those rows carry constraints an implementer must not smooth over: - -- **Hermes reports activation, never completion.** The shipped observer fires - on `action == "loaded"` for a named skill and derives the candidate skill - directory locally. There is no terminal signal in that event, so a Hermes - profile declares `events: ["resource.activated"]` and the reporter must never - synthesize `completed` or `failed` from it. Hermes observation is also - user-scoped until the harness config becomes scope-aware; a repository-scoped - Hermes projection resolves to `unsupported`, not to a user-scoped write. -- **Hermes needs a different runtime binding.** Its plugin is Python, not a - hook command line, which is why the reporter below is a contract with two - bindings rather than one file. -- **DeepSeek Harness is respected, not advertised.** Its `agent/*` event - vocabulary and scoped listeners could carry a real lifecycle observer, and - the plugin is the delivery vehicle for it. It stays `explicit_report` in this - phase because the harness is a developer preview that announces breaking - changes and the adapter pins one exact release (`0.1.0-rc.6`), fail-closed. - Promoting it to `hook` requires pinning a release range and recording a - fixture first, in that order — the same bar as any other client, and the - reason its evidence is not gate-required here. - -## Product contract - -The publisher configures Logion once for a canonical resource version. Logion -produces a **projection directory**, which is a distribution of one resource -version — not a duplicate resource and not a Logion-owned wrapper. - -```text -/ # installed as `owner/repo` -├── plugin.json # portable core, Agent Plugins 1.0.0 -├── skills//SKILL.md # the publisher's artifact, byte-identical -├── .logion/ -│ ├── instrumentation.json # the profile (below) -│ ├── capability.json # the resolved tier and why -│ ├── consent.json # written at install/first activation only -│ └── reporter/report.mjs # the reporter, dependency-free -└── / # hook entry, only for tier `hook` -``` - -Rules the generator enforces: - -- The portable core is copied, never rewritten. A digest comparison proves the - publisher's artifact is byte-identical inside the projection. -- Every emitted receipt names the original publisher and the exact version. -- The projection carries its own `distribution_digest` and `integration_version` - so a fault in the reporter is distinguishable from a fault in the resource. -- `logion instrument` produces a reviewable plan and diff before writing. It - never publishes a package, widens permissions, or enables network delivery - without explicit publisher approval. - -Publisher-side authoring command: - -```bash -logion instrument RESOURCE_VERSION \ - --targets agent-plugin,static-skill \ - --events activated,completed,failed \ - --dry-run -``` - -The publisher has the Logion CLI. The end user does not. That asymmetry is the -whole point of the phase and is the reason `logion instrument` lives in the -CLI while the reporter does not depend on it. - -## The instrumentation profile - -One versioned, Logion-owned file, separate from AI Catalog, ARD, Agent Skills, -and any plugin format. It is the only thing the reporter reads. - -```json -{ - "schema": "logion.instrumentation/v1", - "subject": { - "resource_id": "urn:air:example.com:skill:review-helper", - "resource_version": "1.4.2", - "distribution_digest": "sha256:..." - }, - "publisher": { "identity": "did:web:example.com" }, - "delivery": { - "endpoint": "https://api.logion.sh/v1/resources/RESOURCE_UUID/versions/VERSION_UUID/publisher-receipts", - "mode": "asynchronous-batch", - "max_batch": 20, - "max_spool_bytes": 262144 - }, - "events": ["resource.activated", "resource.use.completed", "resource.use.failed"], - "fields": [ - "resource_id", "resource_version", "distribution_digest", "event", - "outcome", "duration_bucket", "harness", "integration_version" - ], - "excluded": [ - "prompt", "file_content", "local_path", "tool_arguments", "tool_results", - "model_context", "secrets", "user_identity" - ], - "integration_version": "logion.publisher-reporter.v1" -} -``` - -Implementation requirements for the schema, all fail-closed: - -- bounded enums for `event`, `outcome`, and `duration_bucket`; reuse the exact - vocabularies already shipped in `packages/cli/cli/usage/observations.py` - rather than defining a second set; -- size limits per field and per payload; -- canonical serialization (sorted keys, no insignificant whitespace) so the - profile digest is reproducible; -- unknown top-level and unknown field-name rejection; -- an endpoint policy: HTTPS only, one host, no redirects followed; -- a validator plus a `--diff` mode that shows what changed between two profile - versions and whether the change widens data categories. - -The schema, its fixtures, its validator, and the reporter live in -`packages/instrumentation/` in the public repository. Nothing in that package -may import the CLI. The generator resolves `delivery.endpoint` to concrete -identifiers at generation time — a profile shipped with a template in it is -invalid, because the reporter must not build URLs. - -## The reporter - -One contract, two runtime bindings, because the clients do not agree on a -runtime: - -| Binding | Artifact | Used by | -| --- | --- | --- | -| Node | `.logion/reporter/report.mjs`, dependency-free ES module | Agent Plugins clients (Claude Code, Codex) | -| Python | `.logion/reporter/report.py`, standard library only | Hermes, whose plugin API is Python | - -Node is available for the first binding because the ecosystem's own installer -is Node, and Python for the second because the harness plugin runs in-process; -both are assumptions to be **checked, not trusted** — a missing runtime -resolves the tier to `unsupported`. - -Both bindings read the same profile, compute the same digests, emit the same -event shape, and are covered by one shared conformance suite in -`packages/instrumentation/tests/conformance/`. The suite is the contract: a -third binding is added by making it pass, and a binding that diverges on any -case fails the build rather than shipping a second dialect. - -Required behavior, in order, identical in every binding: - -1. Read the hook payload from stdin, bounded to 1 MiB. On any parse failure, - exit 0 silently. -2. Resolve `.logion/consent.json`. If it is absent, or its mode is `off`, or - `DO_NOT_TRACK`/`LOGION_DO_NOT_TRACK` is set to anything outside - `{"", "0", "false", "no", "off"}`, exit 0 and write nothing. -3. Redact before persistence, not before upload: build the event from the - profile's `fields` allowlist only, dropping every other key. A field not in - the allowlist never enters memory as part of the event. -4. Append to a bounded local spool next to the projection. When the spool - reaches `max_spool_bytes`, drop the oldest batch and record the drop count — - never grow without limit and never block. -5. Under `local-only`, stop here. -6. Under `allow`, batch asynchronously with bounded retries and exponential - backoff to the profile endpoint, verifying TLS. Deduplicate by - `(event_id, installation_id)`. -7. Exit 0 always, in under one second of wall clock on the calling path. The - resource's behavior must not change whether the reporter succeeds, fails, or - is absent. - -Prohibitions, each of which needs a test: - -- never install, download, or exec the Logion CLI or any other binary; -- never accept publisher-supplied credentials for Logion; -- never create a stable identifier that correlates a user across resources; -- never infer use from installation, listing, availability, or context loading; -- never emit a terminal outcome the client did not report — absent terminal - signal is `unknown` or no terminal event at all. - -Each binding also exposes, as plain subcommands on the same file: -`status`, `pending`, `export`, `delete`, `disable`. A user who never installs -the Logion CLI must still be able to see, export, and erase everything held -locally, and to turn it off. - -One hook may observe several resources from the same publisher package, but -every event still resolves to exactly one distribution and one -`ResourceVersion`. - -## Capability model and the drift gate - -`capability.json` is generated, never hand-written: - -```json -{ - "tier": "hook", - "client": "claude-code", - "pinned_release": "...", - "hook_contract_fixture": "packages/instrumentation/fixtures/claude-code/post-tool-use.json", - "reporter_binding": "node", - "reporter_runtime": { "required": "node>=22", "present": true }, - "events": ["resource.activated", "resource.use.completed", "resource.use.failed"], - "reason": null -} -``` - -The Hermes equivalent differs in three fields and nothing else: -`"client": "hermes"`, `"reporter_binding": "python"`, and -`"events": ["resource.activated"]`. A Hermes `capability.json` that claims a -terminal event is invalid, and the validator rejects it. - -At install and at every activation the projection re-resolves the tier. Any of -the following forces `unsupported` with a populated `reason`, and it is a -downgrade only — never a silent fallback to inferred telemetry: - -- the installed client version is outside the pinned range; -- the client's hook config shape no longer matches the recorded fixture; -- the reporter runtime is missing; -- `capability.json` is missing, unparseable, or its digest does not match. - -This is the behavior `files.capability_claims_fail_closed_on_drift` asserts, and -it is the reason the phase can advertise a harness at all. - -## Consent UX - -The manager or client presents the gate during installation where that -lifecycle exists, otherwise on first activation, always **before** the first -observation write or network request. - -```text -This skill uses Logion to send usage metadata: -resource/version, activation, completion/failure, approximate duration, -and harness type. - -It does not send prompts, files, paths, tool inputs/outputs, secrets, -credentials, case content, or identity. - -[Allow] [Local only] [Do not allow] -``` - -The copy is generated from the exact profile, not from a generic "telemetry" -label. It lists every outbound data category, the endpoint and operator, the -retention-policy link, and whether a pseudonymous installation identifier is -used. A profile whose `fields` cannot be rendered into that copy is invalid. - -Consent is stored against the tuple: - -```text -publisher identity -+ canonical resource identity -+ instrumentation-profile digest -+ destination endpoint/operator -+ outbound data categories -+ consent mode and scope -``` - -`consent.json` holds the tuple and its digest locally; the server stores only -the digest. Patch and minor resource updates may reuse consent while the tuple -and the publisher's declared compatibility policy are unchanged. Re-prompt -before the first event when the publisher changes, the endpoint changes, -categories widen, retention materially changes, or a previously local-only -profile enables network delivery. A silent profile expansion is invalid. - -Denial is fail-open for the resource: - -- `off`: no observation file, spool, identifier, or network request; -- `local-only`: local counters/receipts only, inspectable and deletable; -- `allow`: only the approved narrow receipt class may be uploaded; -- ratings, prose feedback, prompts, artifacts, eval inputs/results, and public - evidence publication require their own explicit actions or policies. - -Respect the strongest applicable opt-out, including Logion `off`, -`DO_NOT_TRACK`, and a supported native manager's telemetry-disable setting. -Native-manager consent never silently counts as Logion consent. - -## Receipt and evidence semantics - -Keep these facts separate: - -1. `resource.installed`: manager completed an install. -2. `resource.activated`: a verified client lifecycle event selected or loaded - the resource. -3. `resource.use.completed|failed|abandoned|unknown`: the client exposed a - trustworthy terminal signal. -4. `feedback.submitted`: an agent or user intentionally supplied a report. -5. `eval.completed`: a controlled evaluator produced a separately scoped result. - -Only 2 and 3 are normal-use receipts. Neither becomes a star rating, verified -review, benchmark result, safety claim, or automatic bounty recommendation by -itself. - -Publisher-integrated receipts are labeled -`publisher_integrated_field_observation` and carry issuer and integration -provenance. Aggregation exposes sample size, version coverage, consent mode, -harness coverage, concentration, and known blind spots. Independent runners and -controlled evals remain separate evidence classes. - -## Backend - -A new domain in the private repository: `packages/api/api/publisher_receipts/`, -following `maintainer documentation: api-development-guidelines.md` (controller per use case, -schemas in the controller module, services as classes, repository for -persistence). - -- Operation `submit_publisher_receipt`: - `POST /resources/{resource_id}/versions/{version_id}/publisher-receipts`. -- Migration `0048_publisher_receipts` creates `resource_publisher_receipts` - (`0047` is the current head; take the next free number if it moved). - Do not overload `resource_usage_receipts`: that table has no provenance - columns, so a publisher receipt stored there would be indistinguishable from - a CLI-originated one, and the label this phase requires would be a lie. - Columns beyond the 15.11 receipt shape: - `instrumentation_profile_digest`, `distribution_digest`, - `integration_version`, `publisher_identity`, `observation_class`, - `consent_policy_digest`, and a unique `receipt_digest` for idempotency. -- Server-authoritative, never accepted from the request: `identity_tier`, - `pseudonymous_subject_id`, `publisher_verified`, `consent_policy_digest`. - The reporter runs on the user's machine, so each of these is otherwise a fact - the publisher would be asserting about the user. The phase-integrity policy - enforces this; a request schema that declares any of them fails the audit. -- Anonymous receipts reuse the 15.11 contract: a signed pseudonymous subject, - verified by `resource_feedback/services/pseudonymous_subject.py`, with the - subject id derived server-side. Do not add a second signing scheme. -- Add idempotency, rate limits, retention and deletion, endpoint abuse - controls, aggregation thresholds, and publisher-facing health and scorecard - projections. -- Install, activation, field outcome, feedback, and eval evidence stay distinct - typed records. No write path from a receipt to `resource_feedback`, - `course_reviews`, an eval result, a bounty, or a payment. - -## Protocol boundary - -AI Catalog remains the typed catalog/entry representation. ARD remains -pre-invocation discovery. Native formats remain responsible for install and -execution. AKTP remains the evidence and improvement overlay. - -- Do not add Logion consent, hook, or receipt semantics to an AI Catalog base - object, and do not claim ARD performs instrumentation. -- A Catalog Entry may preserve a namespaced scalar metadata reference to a - publisher-hosted instrumentation profile only where the pinned schema and - conformance suite accept it. The native artifact or plugin profile remains - the runtime authority. -- Agent Skill and plugin metadata is namespaced under that format's extension - rules. Tolerated metadata is not executed metadata. -- If an upstream manager gains a standard permission or lifecycle extension, - prefer a fixture-backed upstream proposal over a Logion-only fork. -- Run `python3 scripts/check_protocol_specs.py` and both relevant conformance - suites before changing catalog publication or discovery behavior. - -## Improvement and bounty path - -Consented receipts may reveal version-specific failure clusters, trigger a -post-use feedback request under the user's policy, populate owner-facing -scorecards and regression candidates, nominate an eval, reproduction, -documentation, adapter, or source change, and contribute to a bounty candidate -after cohort, abuse, deliverability, and acceptance-authority checks. - -They never spend money, publish a bounty, assert a root cause, or claim that a -closed upstream resource can be modified automatically. Follow Phase 15.14's -operator approval and deliverable-authority boundary. - -## Implementation order - -Work top to bottom. Each step is verifiable before the next one starts. - -1. **Profile schema.** `packages/instrumentation/` with the JSON Schema, - fixtures (one valid, one per rejection rule), the validator, the canonical - digest function, and `--diff`. Reuse the enums from - `packages/cli/cli/usage/observations.py`; a grep-pinned test forbids a - second vocabulary. -2. **Client fixtures.** Record a real payload per pinned client under - `packages/instrumentation/fixtures//`, and pin the release range. - Claude Code and Codex already have theirs in - `packages/cli/tests/fixtures/hook_payloads/` — reuse those rather than - recording a second copy. **Hermes has none**: record its skill-load event as - `packages/instrumentation/fixtures/hermes/skill-loaded.json`, taken from a - real Hermes session, not hand-written. That fixture is the gating artifact - for the required Hermes leg. -3. **Reporter.** The conformance suite first, then both bindings against it: - `report.mjs` and `report.py`, each with its subcommands, unit tests for every - numbered behavior and every prohibition in [The reporter](#the-reporter), and - a runtime-absent case. Neither binding may be merged while the suite has a - case it does not cover. -4. **Generator.** `logion instrument` in - `packages/cli/cli/commands/instrument/`: resolve the canonical - `ResourceVersion`, emit the projection tree per target, compute digests, - resolve the tier per client, write `capability.json`, and print the plan and - diff under `--dry-run`. Approval-gated write, zero-write dry run. Targets: - `agent-plugin` (Claude Code, Codex), `hermes-plugin`, `static-skill`, and - `dsh-plugin` — the last emitting the Cordis bundle shape - `@logionsh/dsh-plugin` already uses, at tier `explicit_report`. -5. **Backend.** The `publisher_receipts` domain, migration - `0048_publisher_receipts`, and the SDK resource. Regenerate the OpenAPI - contract and run `make contract-audit`. -6. **Scenario.** The proving-ground scenario below, its fixtures, its queries, - its assertions, and the scripted integration test. -7. **Gate.** Run the scenario twice with a real driver, seal - `artifacts/phase-gates/phase-15.11.1.json`, and record any caveat as a - caveat rather than as a passing claim. - -The gate activates as soon as *any* activation path exists, and -`packages/instrumentation/` is one of them, so **step 1 turns it red**, not -step 5. That is intended: the entry fails closed the moment the implementation -starts, and it must not be loosened to keep a branch green, which is exactly -the reward-hacked completion the check exists to reject. - -The consequence is operational, and it decides how this phase is built. While -the gate is red on `main`, the pre-commit hook fails for every commit in the -repository, including work unrelated to this phase, and a permanently red gate -stops being a signal because a genuinely new failure looks identical to the -expected one. So build this phase on one integration branch: merge each step -into that branch, and merge the branch into `main` only when the scenario and -the sealed evidence make the gate green. - -## Mandatory proving-ground scenario - -Follow [the common real-agent gate](agent-proving-ground-phase-gate.md) and its -scenario implementation checklist. Add -`builtin:phase_15_11_1_publisher_integrated_observation` at -`packages/agent-proving-ground/agent_proving_ground/scenarios/builtin/phase_15_11_1_publisher_integrated_observation.yaml` -with `api_adapter: local-devrig`, `driver_config` pinning -`codex: gpt-5.4-mini`, `claude-code: claude-haiku-4-5`, and -`hermes: glm-5.1` (provider `ollama-cloud`), and no -`scripted`/`mock`/`local-process` agent driver. All three are pinned in the -phase policy, and the audit fails with `PHASE_REAL_DRIVER_CONFIG_MISSING` if -the scenario omits any of them. - -Three actors, isolated homes and workspaces: - -- `publisher` — has the Logion CLI, owns the resource, instruments it. -- `consumer` — **no Logion CLI on PATH or in the workspace**, installs the - projection through the real native manager against a local fixture repo. -- `hermes_consumer` — a Hermes-driven agent, also with no Logion CLI. Hermes - must be the driver for this actor: its observer fires on a skill load inside - a Hermes process, so no other driver can produce that event, and a leg that - replays the payload proves nothing (the same trap 15.11's live-hook assertion - exists to close). -- `operator` — observes API state only. - -Phases: - -1. **Publisher prompt:** "Instrument this exact review skill version for the - clients Logion supports. Generate only supported projections. The end user - must not need the Logion CLI. Show me the exact disclosure and the outbound - fields before publishing." -2. **Consumer prompt:** "Install the publisher's plugin through the native - manager. Allow the disclosed minimum usage metadata, use the skill once, and - show me what Logion received." -3. **Consumer prompt (negative leg):** "Turn the telemetry off and use the - skill again." -4. **Hermes leg (required):** `hermes_consumer` installs the `hermes-plugin` - projection, accepts the disclosure, and uses the skill once inside Hermes. - Assert a real activation receipt from the live plugin, and assert that no - terminal outcome was recorded — Hermes cannot report one, and inventing one - is the failure this leg exists to catch. -5. **Fallback leg:** install the static-skill projection in a client with no - pinned hook. Assert normal skill use, a declared - `publisher_observation_unsupported` capability, and zero events. -6. **Drift leg:** move the installed client's pinned marker out of range, then - activate. Assert the capability downgrades to `unsupported` with a reason, - and that no event is emitted. - -Assertions, all required and none `optional: true`: - -| Assertion | Where it lives | What it reads | -| --- | --- | --- | -| `files.instrumentation_profile_valid` | `assertions/files.py` | the generated profile validates and its digest matches | -| `files.native_projection_digest_matches` | `assertions/files.py` | the portable core inside the projection is byte-identical to the publisher's artifact | -| `files.consent_recorded_before_observation` | `assertions/files.py` | `consent.json` mtime and content precede any spool entry or request | -| `files.no_full_cli_installed` | `assertions/files.py` | no `logion` on the consumer's PATH, home, or workspace | -| `files.resource_works_when_disabled` | `assertions/files.py` | the skill's own output artifact exists in the disabled leg | -| `files.publisher_observation_unsupported_declared` | `assertions/files.py` | `capability.json` says `unsupported` with a reason, and the spool is absent | -| `files.capability_claims_fail_closed_on_drift` | `assertions/files.py` | after the drift leg the tier is `unsupported` and no new event exists | -| `files.hermes_hook_projection_observed` | `assertions/files.py` | the live Hermes plugin produced an activation event for the exact version, with no terminal outcome attached | -| `api.publisher_receipt_exact_resource_version` | `assertions/api.py` | the stored receipt names the exact version, publisher, distribution digest, profile digest, integration version | -| `api.install_not_counted_as_use` | `assertions/api.py` | install produced no activation and no terminal outcome | -| `api.private_payload_absent` | `assertions/api.py` | no excluded category appears in the stored record or the API log | -| `api.disabled_use_zero_receipts` | `assertions/api.py` | zero receipts server-side for the disabled leg, not merely a suppressed upload | -| `api.publisher_receipt_never_rates_or_funds` | `assertions/api.py` | no `resource_feedback`, `course_reviews`, eval, bounty, or ledger row resulted | -| `logs.no_500s` | existing | — | - -Every one of these must also be wired for `local-devrig`: add the observed-effect -queries to `api_adapters/_queries.py` and add each assertion's token (the part -after the dot) to the enumerated set in `api_adapters/local_devrig.py`. The -phase-integrity audit fails with `PHASE_ASSERTION_MOCK_ONLY` for any required -assertion whose token is absent from that file, so a scenario that passes -locally can still fail the gate on this alone. - -**Evidence to retain:** pinned client and manager versions, the plugin -manifest, profile and distribution digests, the disclosure copy, the -consent-policy digest, the redacted receipt, the disabled-leg zero-write proof, -the drift-leg capability downgrade, and the no-500 proof. - -## Acceptance criteria - -Each criterion names the check that proves it. `assertion:` names an assertion -this phase's gate requires in `packages/contract-audit/policy/phase-integrity.yaml`; -`test:` is a test path in either checkout. The auditor rejects a criterion with -no marker and a marker naming an assertion the gate does not require. - -- [ ] `logion instrument` turns one canonical `ResourceVersion` into an Agent - Plugins directory whose portable core is byte-identical to the - publisher's artifact and whose recorded digests match the generated tree. - (proof: assertion:files.native_projection_digest_matches) -- [ ] The instrumentation profile validates against its pinned schema, and one - resource version keeps one identity across the plugin and static - projections while distribution and integration faults stay - distinguishable. - (proof: assertion:files.instrumentation_profile_valid) -- [ ] A fresh user sees the exact generated disclosure and can allow, choose - local-only, or decline before any observation state or network request - exists. - (proof: assertion:files.consent_recorded_before_observation) -- [ ] Accepted receipts reach the API from a machine with no Logion CLI - installed. - (proof: assertion:files.no_full_cli_installed) -- [ ] Declining or disabling telemetry never prevents normal resource use. - (proof: assertion:files.resource_works_when_disabled) -- [ ] A disabled or declined projection produces zero receipts server-side, - not merely a suppressed upload. - (proof: assertion:api.disabled_use_zero_receipts) -- [ ] A projection with no pinned client hook, or no available reporter - runtime, declares `publisher_observation_unsupported`, keeps the resource - fully functional, and emits no event. - (proof: assertion:files.publisher_observation_unsupported_declared) -- [ ] When the pinned client release or hook contract drifts from the recorded - fixture, the capability claim fails closed to `unsupported` instead of - falling back to inferred telemetry. - (proof: assertion:files.capability_claims_fail_closed_on_drift) -- [ ] Hermes is a supported `hook` client: its generated stdlib-only plugin - produces a real activation receipt from a live Hermes session, and no - terminal outcome is ever attached to it. - (proof: assertion:files.hermes_hook_projection_observed) -- [ ] The DeepSeek Harness projection ships as a Cordis bundle at tier - `explicit_report`, reporting use through its `logion_*` tools without - claiming automatic observation. - (proof: deferred:dsh is a developer preview pinned to one exact release, so its projection is built and its live evidence is deliberately not gate-required until a release range is pinned and a payload fixture recorded) -- [ ] Every receipt names the exact `ResourceVersion` plus publisher, - distribution digest, profile digest, and integration version. - (proof: assertion:api.publisher_receipt_exact_resource_version) -- [ ] Install, activation, and terminal outcome stay separate facts: an install - never becomes a use, and a missing terminal signal never becomes a - completion. - (proof: assertion:api.install_not_counted_as_use) -- [ ] No prompt, file content, path, tool argument, tool result, secret, or - user identity reaches the request, the stored record, or the logs. - (proof: assertion:api.private_payload_absent) -- [ ] A publisher receipt can inform scorecards and improvement candidates but - cannot create a rating, a review, an eval result, bounty funding, or a - payment. - (proof: assertion:api.publisher_receipt_never_rates_or_funds) -- [ ] AI Catalog, ARD, native execution, and AKTP evidence semantics remain - separate and pass the pinned protocol integrity gate. - (proof: test:scripts/check_protocol_specs.py) - -## Required tests - -These are unit and integration obligations, not gate assertions. They are the -reason the gate above can be short. - -- Consent: allow, local-only, deny, cancellation, non-interactive install, - profile expansion, publisher change, endpoint change, uninstall, reinstall. -- Denial produces a byte-identical no-telemetry state, and the resource's own - output is unchanged between the allowed and denied legs. -- Privacy canaries for prompt, file content, path, tool arguments, tool - results, secrets, identity, and legal or customer content: none may reach - memory serialization, spool, request, logs, or error reporting. -- Install does not fabricate activation; activation does not fabricate - completion; a missing terminal hook yields `unknown` or no terminal receipt. -- Reporter: offline spool bound, retry and backoff, corruption recovery, - deduplication, retention, deletion, endpoint failure, Logion outage, missing - runtime, and a spool at its size limit. -- Exact publisher, resource, version, and distribution attribution across - updates and across two same-named skills. -- Projection reproducibility: instrumenting the same version twice produces - identical digests. -- Capability fixture drift downgrades the claim rather than falling back. -- Full CLI absent: a supported projection still collects the approved narrow - receipt, and the unsupported static fallback remains functional and honest. -- Receipt ingestion cannot create ratings, evals, public evidence, bounty - funding, or payment without the separate required action. - -## Release and re-verification on completion - -Sealing the gate is not the last step. The phase's central claim is that a user -with no Logion CLI gets receipts, and a run against the working tree cannot -prove that — the tree is exactly what such a user does not have. So the phase -completes against a published artifact. - -1. Cut a development release with the **Release all Logion packages** workflow, - `version: 0.2.0.dev2`, `publish_store: false`. The current line is - `0.2.0.dev1`; the orchestrator bumps the four Python packages itself, so do - not hand-edit `pyproject.toml`. The PEP 440 `.devN` suffix is what routes - every Python publisher to the `testpypi` environment and - `https://test.pypi.org/legacy/`. -2. Expect the npm wrapper to be skipped. That is deliberate: npm prerelease - syntax is not PEP 440, so `@logionsh/cli` takes a separately planned - release rather than a misleading shared version. -3. Install the published build in a clean environment, from TestPyPI with PyPI - as the fallback index — TestPyPI does not mirror third-party dependencies: - - ```bash - uv tool install \ - --index-url https://test.pypi.org/simple/ \ - --extra-index-url https://pypi.org/simple/ \ - 'logion-cli==0.2.0.dev2' - ``` - -4. Re-run the publisher leg (`logion instrument`) with that installed CLI, then - re-run the consumer, Hermes, fallback, and drift legs with no CLI in the - environment at all. Every consumer-side assertion must hold against the - published artifact, not only against the checkout. -5. Record in the phase evidence: the installed version, the TestPyPI file - digests, and which legs ran against the published build. A development build - does not touch `manifest-stable.json` or `manifest-latest.json` and is not a - marketplace-attributed companion version — do not describe it as one. -6. Promote to a stable `0.2.0` only through the normal release flow, after the - dev build has carried a full pass. - -If a consumer leg passes on the checkout and fails on `0.2.0.dev2`, the phase -is not complete. That gap is the packaging bug this step exists to find, and it -is the one bug the proving ground structurally cannot see. - -## Rollout and kill criteria - -1. First-party publisher package, local-only by default. -2. Invite-only external publisher with one verified client. -3. Accepted narrow uploads with inspect, delete, and disable surfaces. -4. A second independently maintained publisher and a second supported client. -5. Upstream proposal for a portable install-consent and observer lifecycle, - fixture-backed, filed against the Agent Plugins specification rather than - against a single client. - -Track consent acceptance and decline, activation-to-receipt integrity, exact -attribution, reporter errors, uninstall success, feedback conversion, and -publisher action on the resulting improvements. Never track task contents. - -Narrow or stop the approach if the reporter materially affects resource -reliability, disclosures cannot be precise, exact attribution falls below the -Phase 15.11 threshold, users cannot effectively disable or delete, or -publishers do not act on the resulting evidence. - -## Out of scope - -Covert telemetry, universal static-Skill execution, TLS interception, prompt or -tool-payload capture, cross-resource identity tracking, a new AI Catalog or ARD -runtime contract, independent-eval claims, automatic ratings, automatic bounty -funding, mandatory Logion CLI installation, or mandatory marketplace -acquisition. diff --git a/plans/phase-15.12-ai-catalog-publication-and-ard-discovery.md b/plans/phase-15.12-ai-catalog-publication-and-ard-discovery.md deleted file mode 100644 index 4bf72acd..00000000 --- a/plans/phase-15.12-ai-catalog-publication-and-ard-discovery.md +++ /dev/null @@ -1,428 +0,0 @@ - - -# Phase 15.12 — AI Catalog publication and ARD discovery - -> **Dogfood — Level 3 (discovery):** Logion publishes a conformant AI Catalog, -> exposes/searches it through ARD, consumes both paths with production adapters, -> and verifies zero duplicate resources after acquisition/feedback already work. -> **After this phase:** AI Catalog is the typed publication/index format; ARD is -> the intent-oriented discovery protocol and registry interface built over AI -> Catalog entries. Existing hub crawlers remain ingestion adapters. -> **Honesty boundary:** discovery proves addressability, not quality, safety, or performance. - -## Mandatory dogfood protocol - -The phase-specific prompt below is implementation work, not optional documentation. The implementing agent must exercise the interoperable resource loop delivered by 15.10–15.11: - -1. run local recall, then `logion listings search --query "SEARCH_QUERY" --include-indexed --limit 5 --json` only on LOW/NONE; -2. inspect the exact `ResourceVersion`, distributions, evidence, permissions, license, and acquisition plan—not only a Course projection; -3. obtain explicit approval, run `logion resources acquire RESOURCE_ID --version VERSION_ID --scope repo-root --channel auto --dry-run --json`, then acquire through the recommended Logion or native channel; -4. run `logion resources reconcile --scope repo-root --json` and require exact version attribution; -5. use the resource in the normal harness on this phase's real task and verify it appears in `logion usage pending --json`; -6. submit exactly one intentional post-task report: - -```bash -logion feedback submit RESOURCE_ID VERSION_ID \ - --rating 1..5 \ - --usefulness 0.0..5.0 \ - --reliability 0.0..5.0 \ - --tool-safety 0.0..5.0 \ - --token-efficiency 0.0..5.0 \ - --completed-task \ - --task-class TASK_CLASS \ - --body "One or two resource-focused sentences; no private repository data" \ - --json -``` - -Use `--not-completed-task` when appropriate. Record the feedback ID and `course_review_projection` disposition. A native external installation is valid dogfood; Logion must not require reinstalling it. If acquisition, exact attribution, consent, or actual use is absent, record the blocker and **do not submit feedback/review**. Passive observation alone never justifies a rating. - -## Goal - -Adopt the AI Catalog data model and ARD discovery protocol instead of extending -AKTP into catalog publication or search. - -## Normative layering - -```text -AI Catalog - typed/nestable JSON catalog + entries + host/publisher/trust metadata - served at /.well-known/ai-catalog.json or another allowed location - ↓ indexed/consumed by -ARD - pre-invocation discovery protocol and registry APIs - asks “what resource is available for this task?” - ↓ returns AI Catalog entries/references -native protocol - MCP, A2A, Agent Skills, plugin, OpenAPI, model host, or other runtime - -AKTP (Logion proposal) - optional evidence/improvement events linked to the stable resource identity -``` - -AI Catalog and ARD are complementary but distinct specifications. The -implementation must not call an AI Catalog document an “ARD schema”, pretend -ARD replaces AI Catalog, or put Logion evidence semantics into either base -specification. - -## ASM collaboration boundary - -T0 and T1 are complete. The decisions are recorded in -[`asm-logion-collaboration-and-protocol-convergence-gate.md`](asm-logion-collaboration-and-protocol-convergence-gate.md) -under *Decision record — T1*; read that section before writing any identity or -receipt code, and do not re-derive it from -[`#262`](https://github.com/nicolasmelo1/logion/issues/262). - -This phase still does not adopt ASM and still does not require an ASM production -adapter. The base AI Catalog/ARD work does not wait on an external response: the -phase remains artifact-agnostic and outcome 3 ("remain independent") stays valid. - -Four constraints from that record bind this phase's implementation directly: - -1. The referenced artifact's **immutable content digest** anchors - `ResourceVersion`. An AI Catalog `identifier` maps to a `Resource` through an - explicit source mapping; `version` and source revision are provenance only. -2. A selection-descriptor digest — ASM's `manifest_digest` or any equivalent - from another registry — **must not create a `ResourceVersion`**. Mutable - pricing, SLA and risk change that digest while the artifact is unchanged. -3. No upstream `{kind, digest, issuer}` reference goes on the usage receipt. - `SubmitUsageReceiptRequest` is published `additionalProperties: false` and - enforced `extra="forbid"`, so adding one is a 422, not an extension point. - The reference belongs on a later evidence/AKTP artifact, which this phase - does not build. -4. An unsigned upstream receipt must keep a machine-readable - `"verification_status": "unsigned"` and must never be surfaced as a verified - issuer. - -AI Catalog's open `type` and `metadata` mechanisms allow an independently -governed artifact such as ASM to be represented without changing AI Catalog. -ARD may discover or filter such entries without owning their selection -semantics. Logion therefore: - -- preserves unknown artifact types and namespaced metadata under the pinned - specifications; -- records the original publisher/registry/issuer and immutable fetched digest; -- does not promote mutable price, SLA, risk, or quality values into authoritative - Logion fields merely because a registry returned them; -- does not create a Logion selection manifest or map ASM `service_id` into a - second global identity; -- adds no ASM-specific production behavior until Yi and Logion have a written - boundary decision or a later explicit independent decision. - -One manually approved external catalog may be used as a generic unknown-type -fixture. Calling that catalog ASM-compatible, or using it as evidence of a -partnership, is out of scope. - -## Dogfood prompt for the implementing agent - -```text -Implement Phase 15.12 while using a Logion resource about protocol design, JSON -Schema, catalog/indexer design, or interoperability. Start with -`logion recall search "JSON schema protocol catalog interoperability" --limit 5`. -On LOW/NONE, search the store with -`logion listings search --query "JSON schema protocol interoperability" --include-indexed --limit 5 --json`. -Inspect the best exact resource/version and follow the mandatory native-or-Logion -acquisition/reconciliation protocol. -Use the acquired resource to critique the AI Catalog codec and Trust Manifest, -the ARD search/registry adapter, pagination, unknown-field handling, identity -normalization, and both conformance suites. Record the exact suggestions -used in `artifacts/dogfood/phase-15.12.md`. After the implementation passes its -self-crawl, submit one generic `feedback submit` report for the exact resource/version -actually used and record its Course projection disposition. -``` - -## Spec sources and independent version pins - -- **Mandatory local references before coding:** - [`protocol-specs/README.md`](../protocol-specs/README.md), - [`AI Catalog`](../protocol-specs/upstream/ai-catalog/specification/ai-catalog.md), and - [`ARD`](../protocol-specs/upstream/ard/spec/ard.md). The implementing agent must - also inspect the relevant CDDL/OpenAPI/JSON Schema under - `protocol-specs/upstream/ard/spec/schemas/` and run - `python3 scripts/check_protocol_specs.py`. -- AI Catalog documentation/specification: and - . -- ARD documentation/specification: . -- Official client connectors and shared Agent Finder directory: - and - . -- Use the independently pinned upstream commits and hashes in - `protocol-specs/UPSTREAM.lock.json`. Never implement from this plan's - paraphrase or silently follow a newer upstream `main`. -- Put AI Catalog models under - `logion_indexer/ai_catalog/v1_0/` (or pinned current version) and ARD models - under `logion_indexer/ard/v0_9/`; their dispatchers, version negotiation, - errors, and fixture suites remain separate. -- Unknown optional fields follow each specification's must-ignore/extension - rules. Errors distinguish `ai_catalog_version_unsupported` from - `ard_version_unsupported`. - -## Bootstrap discovery through `ard-connectors` - -Implementing ARD codecs without knowing any Agent Finder endpoint produces an -empty discovery product. Logion therefore consumes the upstream -`ards-project/ard-connectors` directory as a versioned **indexer control-plane -source**. - -It is not installed into the Logion CLI, companion, customer harness, or -`~/.agentfinder`. Those client connectors are useful upstream reference -implementations, but Logion's scheduled indexer queries Agent Finders -server-side and exposes resulting resources through its existing catalog/search. - -### Snapshot contract - -Fetch `agent-finders.json` by an immutable GitHub commit, initially validating -the current shape: - -```json -{ - "selected": null, - "finders": [ - { - "id": "github", - "name": "GitHub Agent Finder", - "description": "...", - "search": "https://agentfinder.github.com/api/v1/search", - "mcp": "https://agentfinder.github.com/api/v1/mcp" - } - ] -} -``` - -- `selected` is a connector-side user preference and is ignored by the - indexer. Logion queries every operator-enabled finder; it never silently - chooses the upstream selected value. -- Store upstream repository, commit SHA, file digest, fetched/activated time, - schema version, validation result, and last-good snapshot. -- A scheduled refresh compares the newest upstream commit with last-good. - Added/changed hosts enter `pending_operator_approval`; removed finders stop - new crawls but retain historical source provenance. -- Malformed or unreachable updates never erase last-good. Surface snapshot age - and `fresh|stale|rejected` status. -- Do not vendor endpoint credentials. Finder auth, if later supported, uses - operator-managed secret references outside the snapshot. - -### Agent Finder query contract - -For each enabled finder, use the upstream canonical request: - -```http -POST -Content-Type: application/json - -{"query":{"text":"","filter":{"type":["..."]}}} -``` - -The scheduler starts with explicit query families/resource types rather than an -unbounded crawler. Preserve finder ID, endpoint, snapshot commit/digest, query -text digest, filters, retrieval time, referrals, raw result digest, relevance -score/explanation, and returned AI Catalog entry identity. - -Relevance score means match quality only. It never becomes Logion trust, -safety, review, or evaluation evidence. - -If a response includes referrals to other Agent Finders, record them as -untrusted candidates. Query only after the same endpoint validation and -operator approval used for directory additions; cap referral depth, fanout, -total requests, entries, and bytes. - -## Concrete file plan - -### Public repository - -- Add `logion_indexer/adapters/ai_catalog.py`, - `ai_catalog/{codec,conformance}.py`, versioned models, and official fixtures. -- Add `logion_indexer/adapters/ard.py`, `ard/{client,codec,conformance}.py`, - versioned request/response models, and official ARD fixtures. -- Add `logion_indexer/sources/ard_connectors.py` for immutable snapshot fetch, - schema validation, diff/approval, and last-good activation. -- Add `logion_indexer/sources/agent_finders.py` for bounded multi-finder - scheduling, referral candidates, source receipts, and result normalization. -- Extend `adapters/base.py`, `models.py`, `pipeline.py`, `dedup.py`, and `pusher.py` to emit generic resources from 15.9. -- Add CLI `logion-indexer crawl --adapter ai-catalog --entrypoint URL`, - `validate-ai-catalog FILE|URL`, `search --adapter ard --registry URL`, and - `validate-ard FILE|URL`. -- Add operator CLI `logion-indexer ard-connectors sync|diff|approve|status` - and `agent-finders run --finder ID|all --query-family FAMILY --dry-run`. - These are indexer/operator commands, not customer CLI installation commands. -- Add `logion/packages/client/.../_resources/resources.py` support for source filters if 15.9 did not already include them. - -### Private repository - -- Add `api/ai_catalog/controllers/get_catalog.py`, - `services/build_catalog.py`, response types, and router registration. -- Serve `/.well-known/ai-catalog.json` and any separately pinned conformance - discovery mechanisms. Build entries from `Resource`/`ResourceVersion`; never - serialize `IndexedListing` directly. -- Add the ARD registry/search surface under `api/ard/` only from the pinned ARD - spec. Do not conflate its search responses with the AI Catalog document. -- Persist source snapshot/run metadata and expose admin/operator status without - leaking secrets or allowing arbitrary public URLs. -- Add settings for public origin, page size (hard max 100), enabled resource types, evidence-link feature flag, and cache TTL. -- ETag is a stable digest of the canonical response page. Honor `If-None-Match`; cache must never mix origins or cursors. -- Add an operator self-crawl job handler in `api/jobs/handlers/` only if it fits the existing job runner; otherwise keep it in the indexer deployment. Do not make API requests recursively from request handlers. - -## Identity and ingestion algorithm - -```text -fetch AI Catalog with HTTPS, timeout, size limit, redirects <= 3 -validate AI Catalog media type/specVersion and conformance level -for each catalog page: - verify cursor loop has not occurred - parse entries without executing or downloading weights - normalize type and canonical URI - resolve version digest; quarantine entries without required immutable identity - upsert Resource + Source + Version through the 15.9 service - record source attribution and import outcome -stop at configured page/resource/byte budgets - -optionally query an approved ARD registry: - validate ARD request/response version and scores as registry-supplied metadata - extract returned AI Catalog entries/references - feed them through the same AI Catalog identity/validation path -``` - -The first scheduled run queries the GitHub Agent Finder and Hugging Face -Discover entries supplied by the pinned `agent-finders.json`, subject to -operator approval and health checks. A resource returned by both becomes one -canonical resource with two discovery-source edges, not duplicate listings. - -SSRF policy: reject private, loopback, link-local, metadata-service, non-HTTP(S), credential-bearing URLs, DNS rebinding, and cross-host redirects unless the host is explicitly allowlisted for the self-crawl test. Reuse any existing safe URL helper; do not create a weaker second implementation. - -## API/CLI outputs - -- AI Catalog entries preserve `identifier`, media `type`, `url|data`, host, - publisher, versions/tags, nesting, and Trust Manifest according to the pinned - schema. Never invent an “ARD ID” when the identity is an AI Catalog entry - identifier such as a domain-anchored `urn:air:` value. -- ARD results preserve registry origin, query, filters, returned entry, - registry-supplied relevance score/explanation, and referral metadata. A - relevance score is not trust or Logion evidence. -- Import report exposes `seen`, `created`, `matched`, `new_versions`, `quarantined`, `errors_by_code`, cursor, source, and duration. -- CLI exits 0 only when the document is conformant; partial import with quarantines is a nonzero documented exit unless `--allow-quarantine` is explicit. - -## Tests and fixtures - -- Separate golden encode/decode against pinned AI Catalog and ARD examples. -- Recorded `ard-connectors` fixtures for current shape, unknown fields, - duplicate IDs, invalid/search host change, removal, last-good rollback, stale - snapshot, and operator approval. -- Recorded Agent Finder request/response fixtures for GitHub and Hugging Face; - exact POST body, pagination/referral behavior, relevance-only labeling, - timeout/rate-limit/5xx, malformed entry, and cross-finder deduplication. -- Prove no connector files, Agent Finder preference, or `~/.agentfinder` state - are written by the Logion client/companion. -- AI Catalog Minimal/Discoverable/Trusted conformance, nested catalogs, - `url`/inline `data` exclusivity, media types, Trust Manifest verification, - poisoning/typosquatting cases, and version compatibility. -- ARD request/search/explore/browse behavior required by the pinned revision, - registry referrals/federation limits, and conversion through AI Catalog - entries without identity loss. -- Unknown-field preservation, unsupported major, malformed cursor, cursor loop, duplicate ID, changed digest, missing digest, oversized page, timeout, redirect, and SSRF fixtures. -- Self-publication integration: start local API, fetch/crawl its real AI Catalog - twice, then discover the same entry through its real local ARD service; assert - one canonical resource/version and zero duplicate creation. -- Deterministic E2E uses a local conformant Agent Finder fixture derived from - the upstream contract. An opt-in staging smoke queries at least one live - upstream-listed finder and records availability/schema drift without making - external uptime a deterministic CI dependency. -- Projection regression: old listing/skills search results do not disappear. -- OpenAPI and client generation checks. - -## Rollout/observability - -- Feature flags: `ai_catalog_public`, `ai_catalog_ingestion`, - `ard_discovery`, `ard_connectors_sync`, and per-finder/query-family allowlists. -- Metrics: fetch latency/status, entries/page, quarantines by code, identity conflicts, duplicate rate, self-crawl drift. -- Metrics additionally include snapshot age/diff status, enabled finders, - finder query success/rate limits, referrals pending approval, results and - exact/ambiguous deduplication. Never include user prompts. -- Start with Logion self-crawl and one manually approved external catalog; expand only after seven clean scheduled runs. -- Activate the pinned GitHub/Hugging Face Agent Finders only after a reviewed - dry-run. Expand connectors only from reviewed snapshot diffs. - -## Build - -- Implement the current AI Catalog schema/Trust Manifest and ARD protocol behind - independently versioned codecs. -- Publish Logion's AI Catalog at `/.well-known/ai-catalog.json`. -- Add AI Catalog ingestion plus an ARD discovery adapter using the existing - discover/enrich/validate/mirror/upsert pipeline. -- Synchronize and pin `ard-connectors/agent-finders.json`; query enabled Agent - Finders from the indexer and preserve per-finder discovery provenance. -- Translate Smithery, GitHub skill sources, MCP registries, and Hugging Face metadata into the generic resource model while preserving source attribution. -- Add official conformance fixtures from both specs and a self-crawl/search test - against the locally served Logion catalog and ARD service. -- Expose `logion resources search|get` and SDK equivalents; existing `skills search` remains a filtered projection. - -## Operational dogfood - -Run a scheduled self-crawl in staging. Fail the job on duplicate identity, digest drift without a new version, invalid ARD output, or projection disagreement. - -## Mandatory proving-ground scenario - -Follow [the common real-agent gate](agent-proving-ground-phase-gate.md). Add -`builtin:phase_15_12_ai_catalog_ard_self_crawl`. - -- **Actors/seed:** `node_operator` and clean `consumer`. - `make proving-ground-seed SCENARIO=phase_15_12` publishes one resource absent - from the consumer index and serves the real local AI Catalog plus ARD - service; a local conformant Agent Finder plus pinned `agent-finders.json` - fixture returns one valid, one duplicate-from-two-finders, one referral, and - one malformed record. -- **Operator prompt:** “Publish this node's AI Catalog, add it as a catalog - source, synchronize the official Agent Finder directory, approve the fixture - endpoints, query all enabled finders, crawl twice, then expose/search the same - entries through ARD. Report accepted/rejected entries without touching the - database or installing connectors in the client.” -- **Consumer prompt:** “Find Fixture Linter through ARD, identify the registry - and original AI Catalog publisher/entry, and inspect its canonical source.” -- **Assertions to implement:** `api.ai_catalog_document_valid`, - `api.ai_catalog_conformance_level_valid`, - `api.ard_search_response_valid`, - `api.ard_connectors_snapshot_pinned`, - `api.agent_finders_queried`, - `api.agent_finder_result_provenance_visible`, - `files.client_has_no_ard_connector_install`, - `api.catalog_crawl_completed`, `api.ard_resource_ingested`, - `api.ard_record_rejected`, `api.self_crawl_no_duplicate`, - `api.resource_source_provenance_visible`, - `api.search_filters_by_type_and_source`, - `api.discovery_succeeds_without_aktp`, and - `api.ingested_model_requires_no_asm_schema`. -- **Negative/evidence:** malformed input is quarantined with a stable reason; - crawl two and ARD discovery add zero duplicates. Retain both spec - versions/commits, AI Catalog document, ARD request/response, crawl counters, - canonical identifier/source, fixture digests, and no-500 proof. - -## Acceptance gates - -Each gate names the check that proves it; see `DEFERRED.md` for -what the markers below leave unproven. - -- [ ] A clean client can consume the staging AI Catalog directly and discover the - same resource through the staging ARD service. - (proof: assertion:api.ard_search_response_valid) -- [ ] The indexer synchronizes a pinned upstream `agent-finders.json`, queries - every enabled approved finder, and makes at least one resulting resource - searchable. - (proof: assertion:api.agent_finders_queried) -- [ ] No ARD connector package or finder preference is installed into customer - clients/harnesses. - (proof: assertion:files.client_has_no_ard_connector_install) -- [ ] Cross-finder duplicate results converge on one canonical resource while - preserving all discovery-source edges. - (proof: assertion:api.self_crawl_no_duplicate) -- [ ] Self-crawl is idempotent and emits an auditable import report. - (proof: assertion:api.catalog_crawl_completed) -- [ ] AI Catalog and ARD failures have separate stable error codes and quarantine. - (proof: assertion:api.ard_record_rejected) -- [ ] Search can filter by resource type and source. - (proof: assertion:api.search_filters_by_type_and_source) -- [ ] No AKTP endpoint is required to discover a resource. - (proof: assertion:api.discovery_succeeds_without_aktp) -- [ ] The published/ingested model remains artifact-agnostic: no ASM-specific - schema, selector, receipt, or partnership claim is required for this phase. - (proof: assertion:api.ingested_model_requires_no_asm_schema) -- [ ] The ASM outreach/ownership matrix exists before any later production adapter - is authorized. - (proof: deferred:satisfied by the T1 decision record in the ASM convergence gate; no scenario asserts that a document exists, and this phase authorizes no ASM adapter) diff --git a/plans/phase-15.14.1-local-multi-agent-first-node-foundation.md b/plans/phase-15.14.1-local-multi-agent-first-node-foundation.md index 1e06c83c..e65a5d4b 100644 --- a/plans/phase-15.14.1-local-multi-agent-first-node-foundation.md +++ b/plans/phase-15.14.1-local-multi-agent-first-node-foundation.md @@ -193,7 +193,7 @@ real-agent scenario below. ### Mandatory real-agent scenario -Add `builtin:phase_15_14_1_local_multi_agent_node` and pass it with GPT-5.4-mini +Add `builtin:local_multi_agent_node` and pass it with GPT-5.4-mini or Claude Haiku against the locally running real API. - **Prompt to host Hermes:** “Start the bounded 15.14.1 foundation smoke with @@ -218,29 +218,44 @@ The canonical policy requires exactly these assertions: `api.role_credentials_isolated`, `api.state_survives_restart`, `files.role_cleanup_complete`, and `logs.no_500s`. -- [ ] The documented command starts the local API/devrig plus consumer and +**Status 2026-09-01: implemented and sealed.** Gate +`artifacts/phase-gates/phase-15.14.1.json` records run +`20260901T171013-phase_15_14_1_local_multi_agent_node` (driver `claude-code` / +`claude-haiku-4-5`, adapter `local-devrig`) with `status: passed`, no caveats and +no unsupported assertions. All eight assertions below passed, plus `logs.no_500s` +and `timeline.no_unredacted_secret`. + +**One bound on that claim.** This phase sits in `_RECOMPUTATION_FREEZE` in +`contract_audit/evidence_contract.py` until `FREEZE_REVIEW_BY` (2026-09-30). Its +evidence contract is declared, but the auditor does not yet recompute the verdict +from typed facts for these pairs — it is holding the specification of what the +15.14.1 hooks do not capture yet. So these boxes rest on the run reporting its own +status, one rung below 15.15, whose facts *are* recomputed. Lifting the freeze is +tracked there, not here. + +- [x] The documented command starts the local API/devrig plus consumer and auditor as actual non-root Compose services. (proof: assertion:sandbox.roles_run_non_root) -- [ ] Both role services run a gate-approved real-agent harness inside their +- [x] Both role services run a gate-approved real-agent harness inside their containers, and each harness process invokes the installed Logion CLI. Host-side drivers coordinate but do not count as either role process. (proof: assertion:sandbox.real_harness_uses_logion) -- [ ] Both role services run with the declared CPU, memory, PID, and wall-time +- [x] Both role services run with the declared CPU, memory, PID, and wall-time limits. (proof: assertion:sandbox.role_resource_limits_enforced) -- [ ] Host Hermes can coordinate without mounting host or cross-role homes, +- [x] Host Hermes can coordinate without mounting host or cross-role homes, credentials, spools, workspaces, keychain, or container socket. (proof: assertion:sandbox.cross_volume_canary_unreadable) -- [ ] Consumer and auditor receive distinct credentials; selective reset +- [x] Consumer and auditor receive distinct credentials; selective reset revokes consumer's old credential without invalidating auditor. (proof: assertion:api.role_credentials_isolated) -- [ ] A fresh consumer session sees the fixture in repository `XPTO`, but not +- [x] A fresh consumer session sees the fixture in repository `XPTO`, but not in repository `ABC`, consumer user scope, or auditor scope. (proof: assertion:files.install_scoped_to_repository) -- [ ] Normal stop/start preserves each role's intended state without making it +- [x] Normal stop/start preserves each role's intended state without making it visible to the other role. (proof: assertion:api.state_survives_restart) -- [ ] Explicit consumer reset removes only consumer disposable state while +- [x] Explicit consumer reset removes only consumer disposable state while auditor remains usable. (proof: assertion:files.role_cleanup_complete) diff --git a/plans/phase-15.15-isolated-first-runner-node.md b/plans/phase-15.15-isolated-first-runner-node.md index 268242be..9dbb1411 100644 --- a/plans/phase-15.15-isolated-first-runner-node.md +++ b/plans/phase-15.15-isolated-first-runner-node.md @@ -153,7 +153,7 @@ The coordinator never promises compute it does not own. Jobs declare requirement ## Mandatory proving-ground scenario Follow [the common real-agent gate](agent-proving-ground-phase-gate.md). Add -`builtin:phase_15_15_isolated_runner`. +`builtin:isolated_runner_node`. - **Actors/seed:** host Hermes `node_operator`, plus real isolated `consumer`, `evaluator`, `contributor`, `sponsor`, and `auditor` containers; the runner is @@ -175,20 +175,35 @@ Follow [the common real-agent gate](agent-proving-ground-phase-gate.md). Add ## Acceptance gates +**Status 2026-09-01: implemented and sealed.** Gate +`artifacts/phase-gates/phase-15.15.json` records run +`20260901T171705-phase_15_15_isolated_runner` (driver `claude-code` / +`claude-haiku-4-5`, adapter `local-devrig`) with `status: passed`, no caveats +and no unsupported assertions. All seven assertions this plan requires passed, +plus `logs.no_500s`. The three unchecked criteria below are the ones no check +has been designed for; they are not unbuilt work, and they stay in +`DEFERRED.md` until an assertion exists that could fail. + +The mandatory dogfood protocol was **not** completed. No container/sandboxing +resource was available to recall, acquire and exercise, so no feedback was +submitted — the blocker is recorded in `artifacts/dogfood/phase-15.15.md` in the +public repository, which is what this plan requires when acquisition or real use +is absent. + Each gate names the check that proves it; see `DEFERRED.md` for what the markers below leave unproven. -- [ ] A fresh runner completes a signed fixture job from claim through artifact +- [x] A fresh runner completes a signed fixture job from claim through artifact upload. (proof: assertion:api.runner_job_completed) -- [ ] Cancellation, timeout, lease loss, retry, and duplicate submission are safe. +- [x] Cancellation, timeout, lease loss, retry, and duplicate submission are safe. (proof: assertion:api.runner_job_terminal_once) -- [ ] A malicious fixture cannot read host secrets or mutate the API service. +- [x] A malicious fixture cannot read host secrets or mutate the API service. (proof: assertion:sandbox.canary_not_exfiltrated) -- [ ] Runner receipts bind resource, environment, inputs, outputs, and assertion +- [x] Runner receipts bind resource, environment, inputs, outputs, and assertion results. (proof: assertion:crypto.runner_receipt_valid) -- [ ] Public runner package can be installed and complete the conformance fixture +- [x] Public runner package can be installed and complete the conformance fixture without access to either repository's source tree. (proof: assertion:api.runner_enrolled) - [ ] On an Apple Silicon Mac, the documented Compose stack starts all five roles, diff --git a/plans/phase-16.1-eval-contract-and-reference-runner.md b/plans/phase-16.1-eval-contract-and-reference-runner.md index ba443f2e..421d84bb 100644 --- a/plans/phase-16.1-eval-contract-and-reference-runner.md +++ b/plans/phase-16.1-eval-contract-and-reference-runner.md @@ -54,6 +54,7 @@ and submit feedback after the portable contract passes two identical runs. - Contract media type: `application/vnd.aktp.eval-contract.v1+json`; authoring YAML is normalized to this JSON before hashing. - Required fields: schema version, subject type/digest constraint, archetype, inputs/fixtures by digest, runtime requirements, steps, metric/assertion definitions, budgets, output paths, redaction, determinism class, and evaluator requirement. - Result media type contains contract/subject/environment digests, assertion vector, typed metric values, outcome, artifacts, resource usage, and limitations. +- The environment digest is computed over **closed, named fields — the ordered harness stack (outermost orchestrator first, each layer with id and version), model id and version, and the declared iteration budget** — not over an opaque blob and not over prose in `limitations`. A result belongs to a stack and a budget, not to a bare harness name; see [Why the pair is part of the result's identity](#why-the-pair-is-part-of-the-results-identity) and [A pair is not always a pair](#a-pair-is-not-always-a-pair). Provide complete JSON Schemas plus typed Python models. Closed enums: outcome, assertion operator, metric kind/direction, determinism class. Extension fields live only under `extensions`; arbitrary top-level keys fail. @@ -120,6 +121,110 @@ someone else's artifact and follows unchanged — author contacted first, rebuttal published verbatim, nothing true and reproducible removed on request. +### Rank contracts by disclosed properties, never by a score + +If Logion measures evaluators, the obvious next question is what measures the +measurer, and the regress has to stop by **policy**, not by adding another layer. +`16.5` already stops it: *"AKTP carries evidence; authority is local, +issuer-aware, and policy-versioned"* — the consumer computes the verdict. + +The binding consequence: **never publish an aggregate "eval quality score."** A +single number makes Logion the terminal authority on who is allowed to measure, +which is precisely the position `positioning-and-independence.md` exists to +refuse, and it would be a badge in everything but name. + +Publish the **declared properties** instead, and let a consumer weigh them under +its own policy: + +- is there a control group? is there any statistical test? +- is there a held-out set? is there a contamination check? +- **does the reference solution pass its own benchmark?** +- who has reproduced it, when, and under which model-harness pair? + +This is the disclosure standard [arXiv:2605.23950](https://arxiv.org/abs/2605.23950) +proposes and that nothing in the field implements. It is also the cheapest +subject type to bootstrap: unlike a skill, a benchmark has inbound citations, +replication literature and published critique, so the free evidence layer is +already written by other people. That property is what makes `eval_contract` and +`model` the fastest path to +[`public-answer-surface.md`](public-answer-surface.md)'s rule that the box never +returns an empty screen. + +### Why the pair is part of the result's identity + +Harness-induced variance can exceed model-induced variance to the point of +reversing model rankings, and the same model scores materially differently under +different scaffolds on the same benchmark +([arXiv:2605.23950](https://arxiv.org/abs/2605.23950)). + +So harness and model are not limitations to disclose at the end of a report; they +are part of what the result *is*. Two requirements follow: + +- `logion eval compare BASE CANDIDATE` **fails closed** when the pair differs, + rather than rendering the comparison with a caveat. +- The version-over-version chart required by `release-0.2.md` must refuse to plot + two points across differing pairs without saying so on the chart. Otherwise a + harness upgrade renders as an artifact improvement, on the most important + public visual the project has. + +Ranking **models** is out of scope and stays out: it is the most expensive thing +available and the most saturated (Artificial Analysis, LMArena, HELM, OpenRouter), +and `release-0.2.md` already fixes models as metadata-first. If model ranking ever +enters, it enters as **model-harness pair** ranking. + +**Correction, 2026-08-27.** An earlier version of this section claimed nobody +publishes pair rankings. That is wrong: Warp Factories advertises benchmark +comparisons across models *and* harnesses, orchestrating third-party harnesses +by name (`claude code`, `codex`, `cursor`, "any MCP-capable coding agent"). The +distinction that survives is the subject, not the axis — Warp's own page says +those benchmarks measure "your factory's agents executing your specific +workflows, **not third-party artifact quality**", and the results stay private +to the customer. What is unoccupied is a **public, reproducible pair-qualified +measurement about an artifact its publisher does not control**, which is what +the observation envelope's first-class `harness` field and the closed environment +fields above are for. See +[`positioning-and-independence.md`](positioning-and-independence.md). + +### A pair is not always a pair + +**Added 2026-09-02.** The section above is right that harness and model are part +of the result's identity, and wrong that a *pair* is enough to express it. A +harness can be wrapped by another harness, and the wrapper moves the number more +than most subjects do. + +[arXiv:2609.01481](https://arxiv.org/abs/2609.01481) (Harness-of-Harness) runs +existing coding-agent harnesses inside an outer planner/developer/QA loop that +keeps versioned project state across iterations. Holding the inner harness and +the model constant, it reports an average relative gain of 52.25% (maximum +82.86%) after three iterations across three harness-model pairs; FrontierSWE +dominance rising 44% to 71% at three iterations and 72.67% at ten against a +27.33% baseline; and GameCraft-Bench 49.58 to 71.52 under Codex + GPT-5.5 +(high). Its own pass-controlled comparison — 71.52 against 58.24 at equal passes +and comparable token usage — is the admission that matters here: the honest +comparison had to hold the *budget* fixed, not just the pair. + +Two consequences bind this phase: + +- **The environment is a stack, not a pair.** "Codex + GPT-5.5" and "Codex under + a three-iteration orchestrator + GPT-5.5" declare the same pair and are not + the same environment. A result that records only the innermost harness is true + in every field it carries and wrong in what it attributes: the orchestrator's + gain renders as the subject's. +- **The budget is part of the identity.** The same stack at three iterations and + at ten is two results, and nothing in the schema stops them from being + compared. + +This is the failure the pair rule already refuses, one layer out. It is cheapest +to close here: `release-0.2.md` records that evidence recorded at 0.2 is +permanent, and a result whose environment cannot name the orchestrator is +permanently uninterpretable rather than merely incomplete. + +**What stays out.** A harness stack is not a new subject type. Indexing an +orchestrator as a `Resource`, so that a published measurement whose subject *is* +a harness stack has somewhere to attach, is a real gap and it belongs to +whichever phase owns the citation layer. This phase only has to record the +environment it ran in. + ## Rollout/acceptance additions - Release and pin the contract package before backend deploy. @@ -135,10 +240,296 @@ and reproducible removed on request. - Normalized result with assertion-level outcomes, costs, artifacts, logs, and failure taxonomy. - `logion eval validate|run|inspect` plus SDK types. +## Implementation guide + +Everything above is *why*. This section is *what to type*, in order. It is +binding: where it disagrees with a summary bullet earlier in this file, this +section wins. Do not begin at step 4 because it looks like the interesting part +— each step's "done when" is the precondition of the next. + +### Repository conventions you must obey + +These are enforced, not stylistic. A PR that violates them fails before review. + +- Line length 79. `ruff` config lives in each package's `pyproject.toml`. +- `typing.Any` is banned repo-wide via `flake8-tidy-imports`. Use a + `TypedDict`/dataclass for a known shape, `JsonValue`/`JsonObject` from the + package's `_json` module at a JSON boundary, or a `TypeVar`. +- `max-complexity = 12`. +- The public package must not import from the private repository, ever. The + dependency arrow is one-way: `logion-eval-contract` knows nothing about + `api/`. +- New public packages join the workspace with `[tool.uv.sources]` entries, the + way `packages/runner/pyproject.toml` declares `logion-client`. + +### Step 1 — `packages/eval-contract/`, the shared library + +The only parser, validator, canonicalizer and result-normalizer in the system. +The private API imports a **pinned released version**; it does not reimplement +any of this, and a second implementation of canonicalization is the specific +defect `api.canonical_digest_agrees` exists to catch. + +Create, in the public repository: + +```text +packages/eval-contract/ + pyproject.toml name = "logion-eval-contract" + logion_eval_contract/ + __init__.py + _json.py JsonValue/JsonObject, copied pattern + schema/ + eval-contract.v1.schema.json + eval-result.v1.schema.json + models.py typed models, closed enums + parse.py YAML|JSON -> model, fails closed + canonical.py JCS canonicalization + digest + normalize.py raw run output -> EvalResult + errors.py the five stable error codes + tests/ + fixtures/ the shared fixture release + test_schema_golden.py + test_yaml_json_equivalence.py + test_canonical_digest.py + test_unknown_fields.py + test_path_traversal.py + test_budget_bounds.py + test_metric_unit_direction.py + test_fixture_digest.py + test_extension_roundtrip.py +``` + +**Contract media type:** `application/vnd.aktp.eval-contract.v1+json`. +Authoring YAML is normalized to this JSON *before* hashing, so the digest of a +YAML file and of its JSON normalization are identical. A test must prove that. + +**Required top-level contract fields.** All required; absence is an error, not +a default: + +`schema_version`, `subject` (`{type, digest_constraint}`), `archetype`, +`inputs`, `fixtures` (each by digest), `runtime_requirements`, `steps`, +`metrics`, `assertions`, `budgets`, `outputs`, `redaction`, +`determinism_class`, `evaluator_requirement`. + +**Required top-level result fields:** + +`contract_digest`, `subject_digest`, `environment` (the closed fields below), +`environment_digest`, `assertion_vector`, `metrics` (typed values), `outcome`, +`artifacts`, `resource_usage`, `limitations`. + +`environment` carries the fields themselves and not only their digest. A digest +alone tells `compare` that two results differ and never which field differed, +which is the difference between refusing a comparison and explaining it. + +**Closed enums.** Any value outside these fails validation: + +| Enum | Values | +| --- | --- | +| `outcome` | `passed`, `failed`, `errored`, `skipped` | +| `assertion operator` | `eq`, `ne`, `lt`, `lte`, `gt`, `gte`, `contains`, `matches` | +| `metric kind` | `count`, `ratio`, `duration_ms`, `tokens`, `cost_usd` | +| `metric direction` | `higher_is_better`, `lower_is_better` | +| `determinism_class` | `deterministic`, `seeded`, `nondeterministic` | +| `contract standing` | `unreviewed`, `contested`, `reproduced`, `superseded` | + +**Extension rule.** Extra keys live only under `extensions`. An arbitrary +top-level key fails. This is what `unknown_field_rejected` in the evidence +contract observes. + +**The environment digest is computed over closed, named fields only** — +`harness_stack`, `model_id`, `model_version`, `iteration_budget` — never over an +opaque blob and never over prose in `limitations`. Read +[Why the pair is part of the result's identity](#why-the-pair-is-part-of-the-results-identity) +and [A pair is not always a pair](#a-pair-is-not-always-a-pair) before touching +this; they are the reason the fields are closed. + +`harness_stack` is an ordered list, outermost first, of +`{harness_id, harness_version}`, with at least one entry; the innermost entry is +the harness that actually invoked the subject. `iteration_budget` is how many +planning/coding/testing passes over one subject the outermost layer was +permitted — `1` for a single-shot run, never inferred, and never omitted. + +A runner that cannot name every layer above it **refuses to emit a result** +rather than reporting the innermost harness alone. That refusal is runner-side +and deliberately not a sixth API error code: `api.invalid_eval_rejected` is +keyed by the five codes in [step 4](#step-4--apievals-backend-integration-private-repository), +and widening that set would weaken the assertion rather than strengthen it. + +**Done when:** `logion-eval-contract` builds, its test suite passes, and the +same golden contract expressed as YAML and as JSON produces one identical +digest. + +### Step 2 — `packages/runner/logion_runner/evals/`, the adapter + +Adapts the **existing** 15.15 job/receipt contract. Do not invent a parallel +execution path, and do not copy proving-ground code — 15.15 reuses +`agent_proving_ground/{runner,artifacts,timeline,redaction,assertions}` through +adapters and this follows the same rule. + +What exists already and must be reused as-is: + +- `logion_runner.job.Lease` — carries `contract_digest`, `sandbox_profile`, + `sandbox_profile_digest`, `input_digests`, `limits`, `artifacts`, + `idempotency_key`. +- `logion_runner.receipt_builder.ReceiptInput` — already has `contract_digest`, + `input_digests`, `output_artifacts`, `assertion_vector_digest`, + `environment_fingerprint`, `redactions_applied`. +- `logion_runner.sandbox.profiles` — profile v0, digest-pinned image only. + +So the eval adapter is a *mapping*, not a new runtime: an `EvalContract` +resolves to a `Lease`-shaped job, and an `EvalResult` is normalized from the +outcome the existing receipt already describes. + +**CLI.** Exposed through the `logion-node` entry point in +`packages/runner/pyproject.toml`: + +| Command | Behaviour | Exit code | +| --- | --- | --- | +| `logion eval validate CONTRACT` | offline schema + semantic validation | `0` valid, `2` invalid | +| `logion eval run CONTRACT --subject PATH\|RESOURCE_ID` | resolve, lease, execute | `0` ran, `2` rejected pre-execution, `1` execution error | +| `logion eval inspect-result FILE` | print normalized result | `0`/`2` | +| `logion eval compare BASE CANDIDATE` | compare two results | `0` comparable, `3` **refused** | + +`run` **resolves every input before leasing or executing**, and prints the exact +contract, subject, image and evaluator digests it resolved. A run that leases +first and resolves later cannot satisfy `rejected_before_execution: true`. + +`compare` **fails closed with exit `3`** when the environment differs between +the two results — a differing harness stack, model, or iteration budget. It does +not render the comparison with a caveat. This is not a preference; a caveat is +what lets a harness upgrade, or one extra orchestration pass, read as an artifact +improvement. + +**Done when:** two executions of a `deterministic` golden contract produce byte- +identical normalized results, and `compare` exits `3` across a differing stack, a +differing model, and a differing budget. + +### Step 3 — the conversion tool + +Companion deterministic scenarios become eval contracts through a checked-in +tool. **Do not hand-maintain parallel copies** — the whole point is that one +source of truth converts. + +The tool must emit, per conversion: `source_scenario`, `source_assertion_ids`, +`converted_assertion_ids`, `dropped_assertion_count`, `added_assertion_count`, +`conversion_tool_version`. Both counts must be `0`. + +Counting alone is insufficient and the gate reflects that: a converter that +drops one assertion and invents another keeps the count identical, so the +**identity sets** are compared, not their sizes. + +**Done when:** every existing companion deterministic scenario converts with +both counts at `0`, and CI compares converted cases against the originals while +the original scenarios keep working. + +### Step 4 — `api/evals/`, backend integration (private repository) + +Storage and read services for contract blobs/digests and result references. +Bulky fixtures stay object-store artifacts; do not put them in Postgres. + +**Validation on upload and on job creation uses the shared library.** The five +stable error codes, all returning **HTTP 422**: + +`eval_contract_invalid`, `eval_subject_mismatch`, +`eval_requirement_unsupported`, `eval_fixture_digest_mismatch`, +`eval_budget_invalid`. + +Each must reject **before** any job row is created. `api.invalid_eval_rejected` +is keyed by these five codes, so a run that exercises four of them does not +close the criterion. + +**Contracts are immutable by digest.** A friendly name may point at a newer +contract; historical runs keep the old digest. There is no update-in-place path. + +**`eval_contract` is a resource type, not a migration.** +`api/resources/constants/resource_types.py` accepts unknown values for forward +compatibility and the column is a plain `VARCHAR(64)`. Adding the constant is +the whole change. A contract is indexed **alongside** the subjects it measures, +in the same index — that is what `indexed_alongside_subject` observes. + +**A result carries the contract digest *and* the contract's standing.** A pass +under a `contested` contract is not a pass under a `reproduced` one, and +rendering them identically is the defect +`api.eval_contract_indexed` exists to catch. + +**Never accept these from the client** on `submit_eval_result` — +`phase-integrity.yaml` fails the build if the request body declares them: +`contract_standing`, `reproduced_by`, `independence_group`, +`environment_verified`, `issuer_id`. A runner reporting its own eval result is +the party with the motive to declare it reproduced. + +**Done when:** upload/get/idempotency/auth/object-store-failure tests pass, +OpenAPI regenerates, and the generated client syncs. + +### Step 5 — the proving-ground scenario + +Add `packages/agent-proving-ground/agent_proving_ground/scenarios/builtin/eval_contract_reference_runner.yaml` +and the assertion handlers. Creating this file **activates the phase**, so the +gate goes red until every assertion below retains the facts its evidence +contract in `packages/contract-audit/policy/phase-integrity.yaml` names. That is +intended; do not create the file first to "see what happens". + +The nine required assertions, their evidence contracts already written: + +`files.eval_contract_valid`, `api.eval_runs_completed`, +`api.eval_result_digest_stable`, `files.eval_reproduced_clean_workspace`, +`api.invalid_eval_rejected`, `files.converted_scenario_assertions_preserved`, +`api.canonical_digest_agrees`, `api.eval_contract_indexed`, `logs.no_500s`. + +Read the `"16.1"` block under `evidence_contracts:` before writing a handler. +It is the list of facts each handler must emit, typed as +`{"ok": true, "value": ...}` or `{"ok": false, "failure": "unreachable"}`. +`status: "passed"` in a report is a claim by the party with the motive and the +auditor does not consult it. + +Every `mutations:` entry in that block names a test in +`packages/contract-audit/tests/unit/test_evidence_contract.py` that **does not +exist yet**. Writing them is part of this phase: each one proves the auditor +actually fails when that fact is wrong. + +**The customer prompt may not name a path.** `customer_fidelity` in the policy +forbids `/Users/`, `/home/`, `${LOGION_PUBLIC_REPO_PATH}`, `uv run`, +`packages/` and `tests/fixtures/` in a goal. A customer reaches the product +through the artifact on their `PATH`; a goal naming the checkout is a replay +recipe, not a customer prompt. + +**Done when:** `make runner-evidence` produces a retained local-devrig run and +`artifacts/phase-gates/phase-16.1.json` seals it with `status: passed`, no +caveats and no unsupported assertions. + +### Order of operations, and what blocks what + +```text +1 eval-contract library ──┬──> 2 runner adapter ──┐ + │ ├──> 5 scenario + gate + ├──> 3 conversion tool ─┤ + └──> 4 api/evals ───────┘ +``` + +Steps 2, 3 and 4 are independent of each other and all depend on step 1. Step 5 +depends on all four. **Release and pin the contract package before the backend +deploy** — step 4 consuming an unreleased step 1 is how the two canonicalization +implementations drift apart without anyone noticing. + +### Things that will look like improvements and are not + +- **Do not publish an aggregate "eval quality score."** Rank by disclosed + properties. A single number makes Logion the terminal authority on who may + measure, which contradicts `16.5` and is a badge in everything but name. +- **Do not make `compare` render across differing environments**, however loud + the caveat. A differing harness stack, model, or iteration budget all refuse. +- **Do not flatten a harness stack to its innermost harness** to make a result + fit a pair-shaped field. Refuse the result instead. +- **Do not reimplement parsing or canonicalization in the private API.** Import + the pinned package. +- **Do not widen `resource_types.py` into a migration.** It is a constant. +- **Do not add a message broker.** v0 polls, exactly as 15.15 does. +- **Do not let the reproduction step see either checkout.** If it can, it proves + the package is redundant rather than sufficient. + ## Mandatory proving-ground scenario Use [the common gate](agent-proving-ground-phase-gate.md) and add -`builtin:phase_16_1_eval_contract`. +`builtin:eval_contract_reference_runner`. - **Prompt/actors:** a creator is told: “Package a portable exact-match eval for this JSON-normalization task, validate it, run it twice on the reference @@ -159,7 +550,7 @@ Each gate names the check that proves it; see `DEFERRED.md` for what the markers below leave unproven. - [ ] Companion deterministic scenarios convert without losing assertions. - (proof: unspecified:the conversion tool has no assertion in the real-agent gate) + (proof: assertion:files.converted_scenario_assertions_preserved) - [ ] Two local executions of deterministic fixtures normalize identically. (proof: assertion:api.eval_result_digest_stable) - [ ] Unsupported requirements fail before execution. @@ -171,8 +562,15 @@ what the markers below leave unproven. (proof: assertion:files.eval_reproduced_clean_workspace) - [ ] The private API and reference runner produce the same canonical digest for every golden contract. - (proof: unspecified:no assertion compares the backend and runner canonical digest for one golden contract) + (proof: assertion:api.canonical_digest_agrees) - [ ] An eval contract is addressable as a resource in the same index as its subjects, and a result carries both the contract digest and that contract's standing. - (proof: unspecified:eval_contract is not yet a known resource type and no assertion covers contract standing on a result) + (proof: assertion:api.eval_contract_indexed) +- [ ] An eval result names its full harness stack, model, and iteration budget + as closed fields, and the environment digest is computed over exactly + those fields. + (proof: unspecified:no assertion inspects the environment field set; api.eval_result_digest_stable proves the digest is stable, not what it is computed over) +- [ ] `logion eval compare` exits `3` when the harness stack, the model, or the + iteration budget differs between base and candidate. + (proof: unspecified:no proving-ground assertion exercises compare across differing environments) diff --git a/plans/phase-16.1.1-a-gate-outlives-its-plan.md b/plans/phase-16.1.1-a-gate-outlives-its-plan.md new file mode 100644 index 00000000..d630f394 --- /dev/null +++ b/plans/phase-16.1.1-a-gate-outlives-its-plan.md @@ -0,0 +1,213 @@ + + +# Phase 16.1.1 — A gate outlives its plan + +A plan is a thing you are about to do. A gate is a thing that stays true. This +workspace welds the two together: every entry in +`packages/contract-audit/policy/phase-integrity.yaml` names a canonical plan, +`phase_integrity.py` reports `canonical plan for phase X is missing` the moment +that file goes, and `evidence_contract.validate()` refuses an +`evidence_contracts` block whose phase has no entry. A delivered plan therefore +cannot be deleted, and [`next-steps.md`](next-steps.md) already says so as a +sequencing rule: *"Closing a phase is not retiring its plan, and from 15.15 on +it usually must not be."* + +So `plans/` has become an archive. Forty-one documents, three of which — +15.14.1, 15.15 and 16.1 — carry work that is delivered and sealed and cannot be +moved, because moving it deletes a check. The one question the directory exists +to answer, what is next, is answered by one table inside a 383-line file, and +every closure makes the file longer. + +## What deleting them actually did + +Five plans were retired the other way, before that rule existed: `15.10`, +`15.10.1`, `15.11`, `15.11.1` and `15.12`, deleted from this workspace on +2026-08-27. All five are still published on the public repository's `main` +today. `plans/` is mirrored there by `scripts/sync_public_roadmap.py`, the +deletion never arrived, and `make public-roadmap-sync-check` names all five +right now as `unexpected mirror file`, next to a stale manifest. + +Deleting a delivered plan did not retire it. It moved the surviving copy to the +one repository where nobody maintains it — published, unlinked and eleven days +stale — while the record of what shipped was compressed into a +`Plan retired 2026-08-27` footnote in the execution order. + +That is the whole argument for this plan in one observation: a retirement that +is half a deletion has no owner, no check and no evidence that it completed. + +## Where a delivered plan goes + +To `maintainer documentation: phases/-.md`: numbered, carrying its acceptance +criteria settled, its exit condition, the sealed gate it produced, and the +paragraph of argument that says what the phase turned out to be about. + +That is already where the shipped shape lives. The retirements the execution +order records point at `../maintainer documentation: ` for what was delivered — native +acquisition, native observation, catalog and ARD discovery — except 15.11.1, +which was never built and points at git history. The change is that the +document *moves* instead of being paraphrased by a footnote, and that it keeps +its number. + +The number is not decoration. The pre-Phase-15 roadmap blocks were deleted on +2026-08-18 for reusing `15`, `16` and `17` for phases that mean something else +today, and nothing in the workspace would notice that happening again. An +archive keyed by number does: a number is spent once, and a spent number +reappearing in `phases:` becomes a finding instead of an archaeology problem. +Sub-numbering inherits the property — `16.1.1` is a child of `16.1`, archives +as its sibling, and cannot quietly become a second `16.1`. That is what makes +the split in [`16.1.2`](phase-16.1.2-a-plan-is-one-pull-request.md) safe to do +repeatedly, and it is why this document is `16.1.1` rather than a slug: it sits +between 16.1 and 16.2, and it spends a number to say so. + +`maintainer documentation: ` is not mirrored to the public repository, so a plan that moves +there leaves it. That is the intended direction, and the mechanism already +exists: the sync unlinks a mirror file whose source is gone. It is the half +that has never run. + +## docs integrity + +`phase-integrity.yaml` governs work in flight. Nothing governs a document after +its work lands, which is exactly why a retirement could sit half-done for +eleven days without a red check anywhere. + +So the archive gets a policy of its own: +`packages/contract-audit/policy/docs-integrity.yaml`, read by +`contract_audit/docs_integrity.py`, scanned from `contract_audit check --mode +fast` next to `phase_integrity.scan`, with its mutations in +`tests/unit/test_docs_integrity.py`, where `make contract-audit-test` runs them +in CI. Six claims, each of them a way retiring a phase has already gone wrong +or could: + +**A number is spent once.** No number appears twice in the archive, and none +appears in the archive and in `phases:` at the same time. + +**A retirement is atomic.** Removing a `phases:` entry without adding the +archived document for its number is a finding, and an archived document whose +number still has a live entry is a finding. Both halves land in one commit or +neither does. + +**The seal outlives the entry.** An archived document names its sealed gate +under `artifacts/phase-gates/` in the public checkout, that file exists, and +the digests it recorded still verify. Retiring an entry retires the obligation; +it must not retire the evidence. + +**Debt does not evaporate.** Every criterion arriving in the archive is `- [x]` +with a dated settlement note, or still carries the `deferred:` / +`unspecified:` marker that `make deferrals` indexes. Closing a phase must not +become a way to empty `DEFERRED.md`. + +**A standing invariant is re-homed before its phase goes.** +`server_authoritative_request_fields` is checked only for *activated* phases, +and 15.15 and 16.1 are the two entries carrying it. Removing those entries +removes a guard on live code that nobody would think to re-add. So the policy +grows a `standing:` block keyed by archived document rather than by phase, and +an entry carrying a standing key cannot retire until that key has a home there. + +**A document lives in one place.** A file present under both `plans/` and the +archive is a finding — the check the five stale mirror files needed and did not +have. + +## What lands here, and what does not + +The machinery, plus the one retirement that is already overdue. The five +2026-08-27 numbers get archive documents; because their plan text is only in +git history, each is a short numbered record naming what shipped, where the +shipped shape lives, and its sealed gate where one exists — `15.11` and `15.12` +have one, `15.11.1` was never built and its record says so. Spending the number +is the point; restoring 2,000 lines of superseded design is not. + +The four live phases stay where they are. 15.14.1, 15.15, 16.1 and 16.2 are +still being read as evidence contracts, and whether a live phase's plan moves +at its own closure or at the release that contains it is a question the first +closure under this machinery should answer, not this document. + +`L3` in the workspace's `sf` policy is untouched here. +`phase-integrity.yaml` remains the phase gate, for the reason its own comment +gives — two gates for one thing is worse than one. Which `sf` rules turn on, +and where, is [16.1.2](phase-16.1.2-a-plan-is-one-pull-request.md). + +## Numbered, and deliberately ungated + +16.1.1, 16.1.2 and 16.1.3 are numbered because they are work between 16.1 and +16.2 and the execution order should say so. None of them becomes an entry in +`phase-integrity.yaml`. + +A phase entry demands a named builtin scenario, a pinned real driver and a +sealed manifest, because a phase is a claim that something shaped like a +customer reached an effect. A policy change and a generator have no customer +effect to reach, and inventing a scenario so that one could be sealed is the +reward-hacking shape this workspace exists to refuse. Their proof kind is +`test:`, which is what that marker is for. + +Which makes the archive's claim stronger rather than weaker: **a number is +spent once, whether or not it ever carried a gate.** `docs-integrity.yaml` is +keyed by number, not by the presence of a `phases:` entry, so these three +retire into `maintainer documentation: phases/` on exactly the same terms as 15.12 did. + +## Acceptance criteria + +- [x] Removing a `phases:` entry without its archived document turns the audit + red, and so does an archived document whose number still has a live entry + (settled 2026-09-07: the `spent:` ledger makes the atomicity claim hold + for every number the workspace consumed, not only the two carrying a + standing invariant — dropping 15.14.1 or 16.2 from `phases:` now goes red) + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [x] A number appearing twice in the archive, or in both the archive and + `phases:`, is a finding + (settled 2026-09-07) + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [x] An archived document whose sealed gate is missing, or whose recorded + digest no longer verifies, is a finding + (settled 2026-09-07) + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [x] A criterion arriving in the archive neither settled with a date nor + carrying its debt marker is a finding, and the debt an archived document + carries is still listed by `make deferrals` + (settled 2026-09-07: the archive is a third source in + `deferral_ledger.collect`, so an archived `deferred:`/`unspecified:` + marker keeps its row in DEFERRED.md) + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [x] A phase entry carrying `server_authoritative_request_fields` cannot + retire until a `standing:` entry names the document that inherited it + (settled 2026-09-07) + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [x] A document present under both `plans/` and `maintainer documentation: phases/` is a + finding + (settled 2026-09-07) + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [x] `maintainer documentation: phases/` holds a numbered record for 15.10, 15.10.1, 15.11, + 15.11.1 and 15.12, each naming what shipped, where the shipped shape + lives, and its sealed gate where one exists + (settled 2026-09-07) + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [x] The `shared_docs` scope in `doc-denylist.yaml` covers + `maintainer documentation: **/*.md`, so the archive is scanned by the same leak checks + as the rest of the shared documentation + (settled 2026-09-07: also fixed the glob rooting — the denylist scanned + from `public_repo.parent`, one directory above the workspace, where the + globs matched nothing) + (proof: test:packages/contract-audit/tests/unit/test_security.py) +- [x] A number with no `phases:` entry still archives, and re-using it is + still a finding, so 16.1.1-16.1.3 are covered by the same claim as a + gated phase + (settled 2026-09-07: the test archives 16.1.1 — no phase entry, no gate — + records it spent, and requires green plus a finding on re-entry) + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [ ] The five retired plans are gone from the public repository and + `make public-roadmap-sync-check` is clean + (proof: unspecified:the mirror is a property of two checkouts and one + merged sync pull request, not of a run in this one — and the sync + workflow itself is red for a reason outside this diff: the missing + PUBLIC_ROADMAP_TOKEN secret and its guardrail step calling two scripts + the public repository deleted in #267. The workflow file is fixed here; + the secret is Nico's to create) +- [ ] Whether a live phase's plan moves at its closure or at the release that + contains it is decided by the first closure under this machinery + (proof: unspecified:16.1 sealed on 2026-09-07 and its evidence contract + is still being read; deciding now is deciding without the case) + +**Exit condition:** `make contract-audit-fast` runs `docs_integrity` beside +`phase_integrity`; `maintainer documentation: phases/` carries a numbered record for every +retired phase; removing a `phases:` entry without one turns the audit red; the +five 2026-08-27 files are gone from the public repository; and `plans/` holds +one fewer document every time something ships instead of one more. diff --git a/plans/phase-16.1.2-a-plan-is-one-pull-request.md b/plans/phase-16.1.2-a-plan-is-one-pull-request.md new file mode 100644 index 00000000..cbc16a61 --- /dev/null +++ b/plans/phase-16.1.2-a-plan-is-one-pull-request.md @@ -0,0 +1,208 @@ + + +# Phase 16.1.2 — A plan is one pull request + +Phase 16.1 is one entry in `phase-integrity.yaml`: one canonical plan, one +scenario, nine assertions. Measured on 2026-09-07, the day its gate sealed, +this is what that one entry has cost: + +| Repository | Pull requests | File changes | Insertions | +| --- | --- | --- | --- | +| `logion` | #317 | 95 | 10,655 | +| `backend repository` | #176 | 16 | 1,414 | +| this workspace | #166, #169, #171, #172, #173 | 28 | 2,374 | + +Seven pull requests, 139 file changes, 14,443 inserted lines — and the phase is +not finished. Two of its criteria are still in `DEFERRED.md` +with no check designed, and +[`driven-actor-performs-the-measured-flow.md`](driven-actor-performs-the-measured-flow.md) +exists to reseal it. 15.15 has the same shape: eight assertions, one pull +request in the public repository at 58 files and 9,554 insertions. + +Nothing about that is reviewable. A 10,655-line pull request is read for style +and merged on trust, which is the exact failure the phase gate was built to +prevent, arriving through the door the gate does not watch. + +## Both rules exist, and here is what each reports today + +`sf 0.4.0` carries them: catalog `f4c2b4b783a5`, 38 rules, +`L4.PLAN_PROOF_BUDGET` among them and `L3.GATE_COVERS_THE_PLAN` carrying the +criterion floor as well as the coverage half. Nothing has to ship upstream +first. Measured by enabling each rule against each repository and reverting: + +| | `plans/*.md` in scope | `sf check` runs | `L4.PLAN_PROOF_BUDGET` | `L3.GATE_COVERS_THE_PLAN` | +| --- | --- | --- | --- | --- | +| workspace | 40 | pre-commit | **33 findings** | duplicate | +| `logion` | 43 mirrored | pre-commit **and CI** | **31 findings** | **here** | +| `backend repository` | 0 | pre-commit | inert | inert | + +All 64 are the floor — *"this plan has no acceptance criteria"* — and **not one +plan exceeds the 60% debt budget**, including the three documents of step 5a at +18%, 20% and 18%. The rule that is supposed to be hard to satisfy is already +satisfied everywhere it applies; what is unsatisfied is the floor. + +`logion` reports 31 rather than 33 because its mirror is stale in both +directions: it still carries the five plans of +[16.1.1](phase-16.1.1-a-gate-outlives-its-plan.md), all of which do have +criteria, and it is missing two that do not. + +In `backend repository` the rule is inert and `sf` says so in the words the plan +would otherwise have had to guess: *"scope matches no plan file: it can never +measure a plan's proof budget."* + +One thing is still owed upstream, and it is not either of these. `logion`'s +policy enables an `L3` rule requiring a driven actor to receive a goal — the +one step 5d depends on — and no released catalog has it, so `sf check` reports +that citation in `logion/docs/factory-rules.md` today, in the document that +exists to explain it. That is a finding this plan inherits rather than creates. + +A note on what is not in scope here: `cargo install --git ... --locked` with no +tag tracks whatever `main` is, so the catalog these two rules come from is not +pinned anywhere. 0.4.0 ships the rule for exactly that +(`L2.CATALOG_ONLY_TIGHTENS`, which compares the running catalog against a +committed fingerprint). It is the obvious follow-on and it is not this plan. + +## What the two rules do, and what they do not + +Neither `L4.PLAN_PROOF_BUDGET` nor `L3.GATE_COVERS_THE_PLAN` measures a diff. +No rule can: the diff does not exist when the plan is written, which is when +the decision that produces it is made. What they do is make a *split* honest, +which is the only lever available before the code exists. + +`L4.PLAN_PROOF_BUDGET` is the floor. It requires a plan to carry at least one +acceptance criterion and keeps the share marked `deferred:` or `unspecified:` +under a budget. Without it, splitting 16.1 into three children is trivially +gamed by giving two of them nothing to prove — they would look as green as the +one that did the work. Measured across the 41 documents in `plans/`: **33 carry +no acceptance criterion at all.** Of the eight that do, `15.14` and `16.8` +carry checkboxes with no proof marker on any of them, and `release-0.2` carries +22 checkboxes with markers on 4. The floor is not a hypothetical. + +`L3.GATE_COVERS_THE_PLAN` is the weld. Every `assertion:` a criterion names +must appear in the gate's `required_assertions`, and a gate that requires +nothing needs a plan with at least one undeferred criterion. Split 16.1 into +16.1.a and 16.1.b and each child's gate has to cover its own promises, or the +split is just a smaller document in front of the same unproven work. + +The thing that forces the split is neither of them, and this plan should not +pretend otherwise: it is a ceiling on what one entry may demand. + +## The ceiling + +One phase entry, one scenario, and **at most five assertions**. Above that, the +phase splits into numbered children, each of which seals on its own. + +Assertion count is the right unit because it is what makes a phase +all-or-nothing. Nine assertions in one gate is nine distinct observable effects +that must *all* land before anything can be sealed, so the pull request cannot +be smaller than the whole phase. Three children of three assertions each seal +three times, and the third one landing late costs the first two nothing. + +Activation breadth is the wrong unit, and the data says so: 16.1 declares three +public activation prefixes and produced 10,655 insertions, while 15.14.1 +declares six and produced far less. Plan length is the wrong unit too, for the +reason the catalog already gives for measuring ratio instead of markdown — it +measures how much somebody wrote. + +Five is a judgement, not a measurement. The four live entries carry nine, +eight, nine and six, so a ceiling of five would have split every one of them, +which is the intended effect. What is honest to say is that no closure has yet +happened *under* this ceiling, so the number carries a review date and the +first two closures after it either confirm it or move it. + +## Why the table reads that way + +The answer differs per repository because `sf check` does not run in the same +places and the three do not hold the same files. + +**`L4.PLAN_PROOF_BUDGET` goes on in the workspace and in `logion`.** The +workspace holds the canonical plans; `logion` holds the mirror and is the only +one of the three that runs `sf check` in continuous integration, on the same +pull requests where the implementation lands. Both are needed: the workspace +copy is where a plan is written, the public copy is where a reviewer meets it. + +It stays **off in `backend repository`**, and the reason is written in that +repository's `docs/factory-rules.md` rather than implied. Its `plans/` holds +`next-steps.md` and nothing else, the rule excludes the execution order by +default, and `L5.NO_INERT_RULE` reports the result. A rule that cannot fire has +to say so rather than read as protection, which is the argument +`L6.DATA_RACES_ARE_DETECTED` already carries in `logion`. + +**`L3.GATE_COVERS_THE_PLAN` goes on in `logion` only.** It needs a gate that +names a plan, and `logion` is the one repository where every part of that claim +resolves: the mirrored plan under `plans/`, the sealed manifest under +`artifacts/phase-gates/`, the assertion handlers under +`packages/agent-proving-ground/`, and a workflow that runs `sf check` on every +pull request. In the workspace it would report the same fault as +`PHASE_GATE_OMITS_PLAN_ASSERTION` at the same boundary — the pre-commit hook — +and two reports of one fault is what the workspace policy already refuses. In +`backend repository` there is no plan and no manifest to point at. + +That makes the public policy a **projection** of `phase-integrity.yaml`, in the +same shape as the OpenAPI contract: canonical on one side, projected to the +other, and the projection's fidelity checked from the only place that can see +both. So `contract_audit` grows one claim — the `gates:` block in the public +policy declares exactly the assertions the canonical phase entry requires — and +the projection is generated rather than retyped, by +[16.1.3](phase-16.1.3-the-documentation-is-read-out-of-the-code.md). + +## What enabling a rule actually costs here + +Per rule, per repository, and worth stating because it is why this is three +pull requests and not one: a mutation fixture under +`.software-factory/mutations//` (`L5.EVERY_CHECK_HAS_A_MUTATION_TEST` +requires the directory, `sf verify` requires it to trip), a section of prose in +that repository's rules document (`L4.EVERY_RULE_HAS_A_WHY`), and `sf lock` in +the same commit (`L2.FACTORY_CONFIG_IS_LOCKED`). + +`L4.PLAN_PROOF_BUDGET` also arrives with 33 existing violations here and 31 in +the mirror. Its ratchet is an allowlist, so `sf ratchet` can hold them — but +`L2.NO_PERMANENT_EXCEPTION` requires a review date on every frozen exception, +which is the correct pressure. The 33 are not one problem: most are reference +documents that were never plans — `positioning-and-independence.md`, +`normative-carry-overs.md` and `consumption-adoption-ladder.md` are standing +policy with no work of their own, and the execution order already says so. +Those belong out of the rule's scope by moving out of `plans/`, not in the +allowlist. The ones that are genuinely +plans and genuinely promise nothing checkable are the finding, and they are the +reason the rule is worth its cost. + +## Acceptance criteria + +- [ ] `L4.PLAN_PROOF_BUDGET` is enabled in the workspace and in `logion`, each + with a mutation fixture that trips it under `sf verify` + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [ ] `L4.PLAN_PROOF_BUDGET` is disabled in `backend repository` with the inert + reason written in its rules document, not merely omitted + (proof: unspecified:a disabled rule produces no finding to assert on; + the claim is a property of the policy comment and the diff) +- [ ] `L3.GATE_COVERS_THE_PLAN` is enabled in `logion` against a declared gate + whose `plan` is the mirrored phase plan, and it is not inert + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [ ] Dropping an assertion from the public policy's `gates:` block, while the + canonical phase entry still requires it, turns the audit red + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [ ] A phase entry demanding more than five assertions in one scenario is a + finding, and the ceiling carries a review date + (proof: test:packages/contract-audit/tests/unit/test_phase_integrity.py) +- [ ] A phase split into numbered children keeps every assertion its parent + required, across the children, with none dropped in the split + (proof: test:packages/contract-audit/tests/unit/test_phase_integrity.py) +- [ ] No document in `plans/` is a standing reference rather than work: the + ones the execution order already describes as having no work of their own + have moved to `maintainer documentation: ` + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [ ] Every remaining `L4.PLAN_PROOF_BUDGET` violation is in the ratchet + allowlist with a review date, and the allowlist is smaller than 33 + (proof: unspecified:the allowlist size is a property of the committed + ratchet, and no run can assert what the right size is) +- [ ] `sf verify` is green in all three repositories and `sf check` is green in + the two where these rules are on + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) + +**Exit condition:** a plan that promises nothing turns `sf check` red in the +workspace and in `logion`'s continuous integration; a gate in the public policy +that requires less than its canonical phase entry turns the audit red; a phase +entry demanding more than five assertions is a finding; and the next phase +after 16.2 lands as numbered children whose largest pull request a reviewer can +read in one sitting. diff --git a/plans/phase-16.1.3-the-documentation-is-read-out-of-the-code.md b/plans/phase-16.1.3-the-documentation-is-read-out-of-the-code.md new file mode 100644 index 00000000..93734636 --- /dev/null +++ b/plans/phase-16.1.3-the-documentation-is-read-out-of-the-code.md @@ -0,0 +1,183 @@ + + +# Phase 16.1.3 — The documentation is read out of the code + +One sentence, borrowed from a repository where it is already load-bearing: + +> Anything a machine can read out of the code is never written by hand. + +## Where this workspace stands + +One generated document. `DEFERRED.md` is 34 lines, produced +by `make deferrals` from the plans' `(proof: ...)` markers, carrying +`Generated by make deferrals. Do not edit by hand.` at the top, and the audit +fails when the committed ledger and the plans disagree. That is the whole +convention, and it works. + +One projection. `scripts/sync_public_roadmap.py` renders `plans/`, +`future-roadmap/` and `protocol-specs/` into the public repository with a +SHA-256 manifest and a `--check` mode that writes nothing. It is red today, for +the reason [16.1.1](phase-16.1.1-a-gate-outlives-its-plan.md) opens with. + +Everything else is by hand: 604 lines of workspace README, 406 in the public +repository, 386 in the backend, 504 in +[`repository-structure.md`](../maintainer documentation: repository-structure.md). + +## What that costs, measured on 2026-09-07 + +**The index that promises completeness reaches less than half.** The workspace +README's *Every README, one index* section opens with "from here you can reach +every repo and package README". It lists 14. Thirty exist. Among the sixteen it +does not reach: `packages/social-management/`, which the same README's third +paragraph instructs every agent to use as the operational surface for anything +public-facing; `packages/eval-contract/` and `packages/runner/`, the two +libraries phases 16.1 and 15.15 delivered; `packages/indexer/`, which is the +engine behind step 2 of the execution order; and +`packages/agent-proving-ground/`, which is the gate. + +**The command list omits the commands the agent instructions require.** The +`Makefile` declares 55 targets. The README documents 32. Missing: `make +install-hooks` and `make deferrals`, both of which `AGENTS.md` tells every +agent to run; the six `pg-*` targets that are the proving ground; the five +`node-*` targets that are what 15.14.1 and 15.15 shipped; and +`make public-roadmap-sync-check`, the check that is currently red. + +Worth stating precisely, because it changes what the fix is: **nothing +documented is missing from the code.** There are no phantom commands and no +phantom READMEs. The drift is one-directional — the documentation lags the +repository, it does not lie about it — which is exactly the drift a generator +removes and a human review does not. + +## Marker blocks, not generated pages + +A generated table sits between `` and +`` on a page that is otherwise somebody's writing. +Everything outside the markers is never touched; everything inside is replaced. +That is what lets an argued page carry a generated index without becoming a +generated page — and it is why `L2.GENERATED_FILES_ARE_LOCKED` stays disabled +here rather than turning on: a hash lock is at file granularity, and these +files are supposed to change by hand outside their markers. Its policy comment +gets corrected to say that, instead of the current "nothing is generated at the +workspace level", which this plan makes false. + +Two cases are errors rather than warnings. A page naming a block nothing +produces, and a block nothing places. A reference section that quietly renders +nothing is worse than a failing build, and a fact the documentation holds and +never shows is the same failure as one that is out of date. + +`make docs-check` writes nothing, exits nonzero, lists the pages that would +change and names the command that fixes them. Nothing generated carries a +clock, a duration, a run count, or a digest of anything outside the +repository — a generated file that changes on every run cannot be checked for +drift. + +## What is derived, and from where + +| Fact | Read from | How | +| :-- | :-- | :-- | +| Every README in the three repositories | the three checkouts | the directory tree, minus vendor and cache directories; the index becomes one block, shared by all three front doors | +| Every workspace command and what it is for | the `Makefile` | the `.PHONY` list, with the comment already written above each target | +| Every shared document and its subject | `maintainer documentation: ` | title and first sentence | +| The packages each repository ships | `packages/*/` and the workspace members | counted, listed, each linked to its own README | +| The CLI's command surface | the CLI parser | `load_cli.py` already walks it for the contract audit | +| The API's endpoint surface | the exported OpenAPI contract | `load_openapi.py` already reads it | +| The plan board: live, archived, next | `plans/`, `maintainer documentation: phases/`, `phase-integrity.yaml` | the phase index in [`next-steps.md`](next-steps.md) is this table, by hand, today | +| The sealed phase gates and what each proved | `artifacts/phase-gates/*.json` | phase, scenario, status, driver, model — never a timestamp | +| Deferred and unproven criteria | the plans' proof markers | already `make deferrals`; it becomes a block instead of a file | +| Which rules each repository enforces, on, off, or frozen, and why | the three `policy.yaml` and `ratchet.yaml` | `sf docs` already generates the rules documents; the counts in prose are not from it | +| The public policy's `gates:` block | `phase-integrity.yaml` | the projection [16.1.2](phase-16.1.2-a-plan-is-one-pull-request.md) needs, generated rather than retyped | + +The last row is the one that matters beyond documentation: it is how the public +repository's gate declaration stops being a second hand-maintained copy of the +canonical phase policy. + +## One shape for three front doors + +The three READMEs read as three different documents today, and the ask is that +they read as one system. So all three take the same four-part shape, and only +the first part is written by hand: + +1. **What this repository is**, and where it sits among the three. Argued, by + hand, short. +2. **The smallest thing you can do to see it work.** One command block. +3. **What is in it** — packages, commands, documents. Every list a block. +4. **Where the full documentation is.** Links out, no summaries. + +Concretely: the workspace README's *Where the product stands* section stays by +hand, because a snapshot of what is true is a judgement and not a fact a +machine can read. Its 215-line dev-rig walkthrough moves to +[`local-development-and-devrig.md`](../maintainer documentation: local-development-and-devrig.md), +which is the document that is supposed to hold it. Its four index and command +sections become blocks. The public README keeps its argument and moves its +install walkthrough to the installer document that already exists. The backend +README's structure, API surface and status sections are read from the tree and +the exported contract. + +The generator runs from the workspace, because it is the only checkout that can +see all three, and its `--check` is verified where the contract audit already +draws that boundary. `L2.DERIVED_ARTIFACTS_MATCH_THEIR_SOURCE` is already +enabled here with `run: make contract-audit-fast`; its scope widens from +`backend repository/packages/api/**` to include the sources a document is read +out of, so the same command fires when one of them moves. One rule, one +command, wider activation — a tightening, which is what +`L2.POLICY_ONLY_TIGHTENS` requires of any change to it. + +## Non-goals + +**No drafting.** Nothing asks an agent for prose about an undocumented surface +and puts the answer in a gate. The blocks are indexes and tables; the argument +stays written by a person. + +**No generated front door.** The README is argued rather than listed, and +generating it whole would cost the one piece of writing that makes anybody read +further. It carries blocks. It is not one. + +**No documentation site.** A navigation manifest is worth generating once +something consumes it, and nothing does. + +## Acceptance criteria + +- [ ] `make docs-check` writes nothing, exits nonzero, names every page that + would change and the command that fixes them + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [ ] A block replaces only what sits between its markers, leaving the rest of + the page byte-identical + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [ ] A page naming a block nothing produces fails, and a block nothing places + fails + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [ ] The README index reaches every README in all three repositories, and + adding a package makes the block change without any list being edited + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [ ] Every `make` target is documented from the `Makefile`, and adding a + target makes the block change + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [ ] Nothing generated carries a value that differs between two runs on the + same commit + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [ ] The public policy's `gates:` block is generated from + `phase-integrity.yaml`, and editing it by hand turns the check red + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [ ] All three READMEs follow the four-part shape, and the workspace README no + longer holds a dev-rig walkthrough that + `local-development-and-devrig.md` is responsible for + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [ ] `L2.DERIVED_ARTIFACTS_MATCH_THEIR_SOURCE` activates on the sources a + document is read out of, not only on the backend API package + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [ ] `make deferrals` writes a block rather than a whole file, and + `DEFERRED.md` either becomes that page or stops existing + (proof: unspecified:whether the ledger stays a page of its own depends on + where it reads best, which is an editorial call this plan should not make + in advance) +- [ ] Whether the three READMEs share one generated index or each carries its + own is decided when the first one is converted + (proof: unspecified:the shared block has to survive the public + repository's link rewriting, which is only observable once one is + rendered) + +**Exit condition:** `make docs-check` runs inside `make contract-audit-fast`, +writes nothing, and goes red when the code moves and a page does not; every +index, command list and count in the three READMEs comes from a block whose +source is named in `maintainer documentation: `; and adding a package, a command or a +document requires no second list to be edited by hand. diff --git a/plans/phase-16.1.4-a-published-entry-resolves.md b/plans/phase-16.1.4-a-published-entry-resolves.md new file mode 100644 index 00000000..afb6d9f2 --- /dev/null +++ b/plans/phase-16.1.4-a-published-entry-resolves.md @@ -0,0 +1,201 @@ + + +# Phase 16.1.4 — A published entry resolves + +On 2026-09-08 an outside scanner scored `www.logion.sh` **72/100** for agent +readiness: `npx ax audit`, the MIT-licensed CLI for ora, whose check set is the +one Vercel's readiness tool runs. The score is not the finding and this plan +does not chase it. Following our own catalog's three entries by hand is. + +## What our own catalog does when an agent follows it + +Measured 2026-09-08, against the live hosts rather than the repository, because +the landing edge is what serves the apex: + +| Entry | URL | Result | +| :-- | :-- | :-- | +| `urn:air:logion.sh:openapi:marketplace-v1` | `api.logion.sh/openapi.json` | 200, 288 KB | +| `urn:air:logion.sh:skill:logion` | `www.logion.sh/.well-known/agent-skills/logion/SKILL.md` | 200, digest matches the one `agent-skills/index.json` publishes | +| `urn:air:logion.sh:catalog:node` | `api.logion.sh/.well-known/ai-catalog.json` | **403 — `{"detail":"ai_catalog.public feature flag is disabled"}`** | + +One entry in three is a dead pointer, and the builder that emits it says why +that is the worst available outcome. From `ai_catalog()` in +`packages/landing/landing/main.py`: + +> Entries are read from site.yaml rather than assembled here, so the catalog +> can only ever advertise an artifact the content file actually declares — a +> catalog entry pointing at nothing is worse for an agent than no catalog at +> all. + +Declaring is not resolving. The docstring states an invariant that nothing +checks, and the guarantee it does offer — that an entry exists in `site.yaml` — +is satisfied by the entry that fails. + +**The design is right and this plan does not change it.** The `site.yaml` +comment argues the case: the apex stays the single discovery entry point and +nests the node catalog rather than duplicating it, because two documents +answering one well-known URI on two hosts would make the apex and the API +disagree about what Logion offers. That reasoning holds. What is wrong is that +the nested pointer names a document gated behind `ai_catalog_public`, which +defaults off in `api/feature_flags/seed.py` and is off in production. + +So there are two implementations of one document: one hand-curated from +`site.yaml` and served by the landing edge, one generated from the resource +layer and served by the API with the switch off. The scanner passed us on +catalog presence, because presence is what it measures. Presence is precisely +the signal Logion exists to say is not evidence, and we shipped the version of +that failure that is ours. + +## The exported contract does not say how to authenticate + +`api.logion.sh/openapi.json`, read 2026-09-08: **117 paths, 46 of them POST, +`components` carries `schemas` and nothing else.** No `securitySchemes`, no +top-level `security`, no per-operation `security`. The API is bearer +authenticated in practice and its published contract declares none of it. + +ora reported this obliquely, as a missing 401 with `WWW-Authenticate` and +missing RFC 9728 metadata. The sharper statement is available from our own +artifact, and `packages/contract-audit/contract_audit/load_openapi.py` already +reads that artifact on every audit. An agent that fetches the contract — the +first entry in our own catalog, the one that resolves — cannot learn how to +authenticate against it. Publishing more well-known documents does not fix +that; the contract admitting its own scheme does, and the documents are worth +adding after it. + +## Ten POSTs move money and one path in the whole API is idempotent + +`idempotency_key` appears in the contract on exactly one path, +`/v1/executions/jobs`, as a body field. These ten POSTs move money or +entitlements and none of them carries an idempotency key in any form: + +```text +/v1/credits/top-ups /v1/bounties +/v1/courses/{course_id}/purchase /v1/bounties/{id}/payouts +/v1/admin/bounties /v1/bounties/{id}/submissions +/v1/admin/bounties/{id}/fund /v1/bounties/{id}/submissions/{id}/open-pr +/v1/admin/bounties/{id}/submissions/{id}/accept +/v1/admin/bounties/{id}/submissions/{id}/reject +``` + +The consumer this API is designed for is an agent, and an agent retries. A +retried `POST /v1/credits/top-ups` after a gateway timeout is a double charge, +and a retried payout is a double payout. Whether the key is an +`Idempotency-Key` header or a body field is a contract decision; that the money +paths have neither is not. + +## The one field Logion should not be leaving empty + +Neither the host block nor any entry carries a `trustManifest`. From the pinned +specification (`protocol-specs/upstream/ai-catalog/specification/ai-catalog.md`, +authoritative over this summary): + +```text +TrustManifest = { identity, ?identityType, ?trustSchema, ?attestations, + ?provenance, ?privacyPolicyUrl, ?termsOfServiceUrl, + ?signature, ?metadata } +Attestation = { type, uri, ?digest, ?size, ?description } +ProvenanceLink= { relation, sourceId, ?sourceDigest, ?registryUri, + ?statementUri, ?signatureRef } +``` + +A company whose product is attestation about artifacts publishes a catalog with +the attestation slot blank. That is worth fixing for the same reason the dead +entry is: it is our own surface making our own argument badly. + +**Filling it is not a trust claim, and the sequencing matters.** `identity`, +`provenance` and the policy URLs are facts about who we are and where the +artifact came from; they go in now. `attestations` is a list of URIs to +findings, and a `trustManifest` whose attestations point at our own marketing +is the self-rating that +[`positioning-and-independence.md`](positioning-and-independence.md) forbids — +the same failure as a store issuing its own score. An attestation entry appears +only when its `uri` resolves to a real, reproducible finding a reader can check +with no account, which is the step 6 artifact, not this plan's. + +On `identityType`: `site.yaml` already argues why the host identifier is a bare +domain rather than `did:web:logion.sh` — resolving a DID requires serving a DID +document we do not publish. ora serves `did:web:ora.ai`. Copying the form to +score a check while the document 404s would be a worse version of the failure +this plan opens with. + +## What is refused, and why it is not a backlog + +Five of ora's failing checks are presence-and-popularity measures: `Listed on +skills.sh`, `Skills.sh skill quality`, `ChatGPT app listed`, `Wikipedia / +Wikidata`, `Agent Plugins manifest`. Two of them are satisfied by publishing +into a competing index. That is the category Logion left, and its own README +records why volume there is lost and irrelevant. They do not enter the backlog, +and a later reader should not rediscover them as oversights. + +Also refused: displaying or storing the 0–100 score anywhere in Logion's +surfaces or documents. A registry's mutable quality value is never promoted +into an authoritative Logion field — the rule already written for ARD-supplied +relevance in +[`../maintainer documentation: ai-catalog-and-ard-discovery.md`](../maintainer documentation: ai-catalog-and-ard-discovery.md) +covers this exactly. + +Not refused, but not here: WebMCP, NLWeb `/ask`, A2A agent card, `auth.md`. +Emerging surfaces with no consumer we currently serve. `Brand name +discoverability` — a search for "Logion" returns eight results without our +domain — is real information about a generic name colliding with older +meanings, and it is not a site fix. + +## One thing this measurement did change + +[`../maintainer documentation: ai-catalog-and-ard-discovery.md`](../maintainer documentation: ai-catalog-and-ard-discovery.md) +records that on 2026-08-17, of eleven announced ARD backers and nine adjacent +vendors, only `huggingface.co` served `/.well-known/ai-catalog.json`. As of +2026-09-08 `ora.ai` serves a valid one: `specVersion 1.0`, +`did:web:ora.ai`, entries including an MCP server card. That is a second +catalog in the wild and a candidate `ard-connectors` source. It is not an +issuer #2 candidate: readiness of a website is a different subject class from +behaviour of a catalogued artifact, and the milestone is not moved by it. + +## Non-goals + +**No `ax` in the measurement path.** It measures a different subject, and +putting a third-party npm package inside the runner or scanner chain would put +someone else's release cadence in the path of the thing we sell. + +**No new well-known document nothing consumes.** The failure here is documents +that do not resolve, and adding more of them is the same mistake. + +## Acceptance criteria + +- [ ] Every catalog entry served by the landing resolves in the test suite, and + an entry whose URL is served by another host is published only with the + check that proves it resolving alongside it + (proof: test:packages/landing/tests/test_landing_agent_discovery.py) +- [ ] `urn:air:logion.sh:catalog:node` either resolves in production or is not + advertised, and the entry cannot be reinstated without the check + (proof: unspecified:the state of `ai_catalog_public` in production is not + observable from any test in either repository; the choice between + enabling the flag and withdrawing the entry is made when it is made) +- [ ] The exported OpenAPI declares a security scheme and every authenticated + operation references one, checked where the audit already loads it + (proof: test:packages/contract-audit/tests/unit/test_load_openapi.py) +- [ ] A 401 from an authenticated path carries `WWW-Authenticate` + (proof: test:packages/contract-audit/tests/unit/test_load_openapi.py) +- [ ] Every POST that moves credits, entitlements, bounty funds or payouts + declares an idempotency key, and adding a money-moving POST without one + turns the audit red + (proof: test:packages/contract-audit/tests/unit/test_load_openapi.py) +- [ ] The host block and every entry carry a `trustManifest` whose `identity` + resolves, generated rather than retyped + (proof: test:packages/landing/tests/test_landing_agent_discovery.py) +- [ ] An `attestations` entry whose `uri` does not resolve to a published + finding fails the build, so the slot cannot be filled with marketing + (proof: test:packages/landing/tests/test_landing_agent_discovery.py) +- [ ] No Logion document or surface carries a third-party readiness score as a + Logion fact + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) +- [ ] The ARD adoption paragraph in `ai-catalog-and-ard-discovery.md` states + what was true at its most recent probe and names the date + (proof: test:packages/contract-audit/tests/unit/test_docs_integrity.py) + +**Exit condition:** every entry in every catalog Logion publishes resolves for +an agent that follows it with no account; the exported contract states how to +authenticate against it; no money-moving POST can be added without an +idempotency key; the trust manifest carries identity and provenance and is +structurally incapable of carrying an attestation that does not resolve; and +the five presence checks are recorded as refused rather than pending. diff --git a/plans/phase-16.1.5-what-is-installed-here-has-a-version.md b/plans/phase-16.1.5-what-is-installed-here-has-a-version.md new file mode 100644 index 00000000..6555d47d --- /dev/null +++ b/plans/phase-16.1.5-what-is-installed-here-has-a-version.md @@ -0,0 +1,231 @@ + + +# Phase 16.1.5 — What is installed here has a version, and its use is recorded against it + +Logion publishes a SHA-256 for its own companion skill at +`/.well-known/agent-skills/index.json`. The copy of that skill installed on the +founder's machine has a different digest, has had one since it was copied in on +2026-06-28, and nothing on the machine knows. + +```text +published sha256:faf7f7f7cf3e42760f3c3decea7be42048c726ce8d9998900d409c441188ef0e +installed sha256:366a74d9854983a9c31b39581985d98ef6f1fb3e6151a858ebbcae12c1016b63 +``` + +The published digest matches `packages/agent-companion/SKILL.md` exactly, so the +landing's drift test is doing its job. The gap is on the other side, where +nothing is watching: the artifact that is loaded into every session. + +## The census, measured 2026-09-08 + +Fourteen skills are installed at `~/.claude/skills/`, all of them loaded into +every session on this machine: + +| Family | Installed | Version on disk | Upstream | State | +| :-- | :-- | :-- | :-- | :-- | +| `logion` | 2026-06-28 | `manifest.json` 0.1.4 | published digest, CLI at 0.1.15 | **stale** | +| `factory-*` (4) | 2026-08-18 | none | `software-factory` `v0.4.0`, tagged 2026-09-04 | **2 of 4 differ from the tag** | +| `amy-*` (7) | 2026-09-03/05 | none | `amy` `main` | 6 match HEAD, which is not a release | +| `amy-develop` | 2026-09-03 | none | **withdrawn upstream** | still installed, still loaded | +| `antikus-code-review` | 2026-07-31 | none | not recorded anywhere | unknown | +| `find-skills` | symlink | none | `~/.agents/skills/` | out of this tree | + +`amy-develop` is the sharpest one. Commit `a850603` in `amy` is titled *"chore(skills): +amy-develop belongs to this repository, not to the install"* — the skill was +deliberately removed from the installable set. It is still on this machine, and +it is still in the skill list of every session started here. Upstream withdrew +it and the machine never heard. + +**One of the fourteen carries a version at all, and it is Logion's own, and it +is wrong.** None carries a digest that anything checks. + +Beyond the skills: + +- **18 `@amykit/plugin-*` packages at 0.2.0**, installed into amy's home and + invisible to `~/.claude/plugins/installed_plugins.json`, which knows exactly + one plugin on this machine: `frontend-slides@2.1.0`. +- **`sf` 0.4.0**, a cargo-installed binary invoked as `sf check`, used in this + workspace and inside amy. + +This is the product's own thesis, at N=1, failing on the founder's own laptop: +nobody publishes what an installed artifact actually is, so nobody can tell you +whether the thing you are running is the thing that was measured. + +## Why the loop 15.11 shipped cannot see any of it + +Four mechanical reasons, all of them findable in the code rather than inferred: + +1. **It has never been switched on here.** `~/.logion/integrations.json` is + `{}`. There is no `usage/` spool and no `inventory/` directory. Consent + defaults to `off` per harness, which is right, and nobody ever said yes. +2. **Attribution is path-based against acquisition receipts.** + `cli/usage/attribution.py` matches path-shaped arguments against + `_receipts.load_receipts()`. With no receipts, every observation resolves to + nothing and is dropped. +3. **A receipt requires a catalog match.** `resources reconcile` records a + receipt only where a locally installed artifact resolves to a *unique + catalog resource*. None of these artifacts is catalogued, so no receipt can + exist, so nothing is attributable. This is the actual blocker and it is one + rule. +4. **`discover_native_state` knows two managers**, `skills` (a + `skills-lock.json`) and `plugins` (`.agents/plugins/manifest.json`, + `.claude/plugins.json`, `plugins.json`). Every artifact above arrived by + directory copy, by npm into amy's home, or by cargo. None of the three files + it reads is in play. + +And a fifth, which is deliberate and should stay deliberate: +`_PATH_TOKEN_RE` requires a separator before a token counts as a path, because +*"a bare word like `pytest` is a program name, not evidence"*. `sf check` is a +bare word. **software-factory cannot be observed today even in principle.** + +## The shape: index what is already being run, then the existing loop works + +The cheap correct move is not a second observation mechanism. It is to make +these artifacts catalogued subjects, because everything downstream — receipts, +attribution, `usage pending`, `feedback submit --rating` against a version — is +already built and already tested, and starts working the moment a subject +exists. + +All three families are public and indexable under the settled +`indexed → improving → claimed` boundary, with attribution, source link and +honest tiering: + +- **amy plugins** — 18 npm packages under `@amykit/`, with versioned releases + and a repository field pointing at the directory each came from; +- **software-factory** — a public repository with tags `v0.2.0`, `v0.3.0`, + `v0.4.0`, and a `catalog/` whose fingerprint `sf` 0.4.0 already compares + against a committed value under `L2.CATALOG_ONLY_TIGHTENS`; +- **the skills** — files, with digests, some already published. + +They are first-party in the sense that matters: the same person wrote them. +That is a disclosure obligation, not a disqualification. They are tiered +`indexed`, the authorship relation is recorded, and **a subject this operator +authored is never the example used to argue that the method is independent.** +Publication of anything from this corpus goes through +[`../maintainer documentation: measurement-publication-playbook.md`](../maintainer documentation: measurement-publication-playbook.md). + +## Version identity, which is not the same fact three times + +- **A skill** is a directory. Identity is its content digest, resolved against + the publisher's digest where one is published — Logion publishes one, `amy` + and `software-factory` do not. Where upstream publishes no digest, identity + is (source repository, tag or commit, path) and the receipt carries + `verification: unverified` rather than implying a check that did not happen. +- **An amy plugin** is `name@version` from npm, and amy's own home records + which are installed. That state file is one adapter in + `discover_native_state`, in the same shape as the two already there. +- **software-factory** is a release tag *plus* the catalog fingerprint. Two + builds at the same tag with different rule catalogs are not the same subject + for anything we would want to say about them, and `sf` already computes that + fingerprint for its own rule. + +The last one carries a prerequisite that belongs to the operator, not to the +code: 16.1.2 records that `cargo install --git ... --locked` with no tag tracks +whatever `main` is. **A subject that is "whatever main was that day" cannot be +measured.** Installing from the tag is a one-line change to the install +command, and it is the whole of what "pinned at the release version" requires. +Nothing inside `software-factory` changes. + +## The bare-word exception, stated as narrowly as it can be + +To observe `sf`, exactly one thing changes: an executable name is attributable +**only when it exactly equals the program name declared by a censused +installation**. No path is inferred, no other token in the command line is +read, the command string still never leaves memory, and an unrecognised program +name is still dropped. That is a lookup against a declared whitelist, not path +inference, so the sentence in `attribution.py` stays true as written: a bare +word is still not evidence — a bare word that matches a declared installation +is. + +## Nothing in amy or software-factory is modified + +This is the constraint in the ask and it is also the correct engineering +position: **an artifact that has to be modified before it can be measured +cannot be measured by a third party**, which would make the whole thesis +circular. The observer is the harness-level `PostToolUse` hook the CLI already +writes through `integrations enable`, plus a census command. amy's plugins, +workflows, gates and packages are untouched. `software-factory` is untouched; +only which build is installed changes. + +## Two profiles, because one of the machines belongs to an employer + +amy runs against `revv` repositories at work. The envelope already carries no +path, prompt, tool argument or free text, and `scope_id` is an HMAC keyed by +the local home rather than a hash of the repository path — so nothing in a +record names a work repository. That is necessary and not sufficient. + +- The **work profile stays `local-only`**: it spools and never uploads. What + may leave that machine is a hand-reviewed aggregate — counts per artifact + version — not per-event receipts. +- The **personal profile runs at `prompt`**, so every upload is an explicit + act. +- `DO_NOT_TRACK` / `LOGION_DO_NOT_TRACK` continue to win over both. + +## Feedback is written after a task, not after an observation + +`logion feedback submit RESOURCE_ID VERSION_ID --rating ... --task-class ...` +already exists and already requires a version. The 15.16 rule holds unchanged: +passive observation never justifies a rating. A report is written after a real +task, names the exact version, and says what happened. + +At N=1 the honest claim is "one operator, N sessions, this version", the report +says so in those words, and a measurement is not an endorsement. That sentence +has to appear beside the finding rather than in a footer, which is the same +rule every other Logion surface is held to. + +## What this is not + +- **Not 15.11.1.** That is observation of *other people's* users, needs + consent, and is off the critical path. This is one operator observing his own + machine. +- **Not telemetry, and not a product surface.** No dashboard ships here. +- **Not a reason to move commerce earlier**, and not an independence claim. + First-party evidence about artifacts the operator wrote is disclosed as + exactly that. + +## Acceptance criteria + +- [ ] `logion inventory census` lists every installed skill, plugin and + declared program with a version identity or an explicit `unknown`, and an + artifact with no upstream is reported rather than skipped + (proof: test:packages/cli/tests/test_inventory_census.py) +- [ ] An installed artifact with no catalog match receives a local receipt + marked `unverified` instead of no receipt at all + (proof: test:packages/cli/tests/test_reconcile_uncatalogued.py) +- [ ] An installed artifact whose digest differs from its upstream's published + digest is reported as drifted, with the stale companion skill on this + machine as the fixture + (proof: test:packages/cli/tests/test_inventory_census.py) +- [ ] An installed artifact whose upstream withdrew it is reported as orphaned + rather than as current, with `amy-develop` as the fixture + (proof: test:packages/cli/tests/test_inventory_census.py) +- [ ] amy's home plugin state is an adapter in `discover_native_state` + alongside `skills` and `plugins`, and adding it changes no existing + adapter's behaviour + (proof: test:packages/cli/tests/test_reconcile_uncatalogued.py) +- [ ] A program name is attributed only on exact match against a censused + installation, and an unknown bare word is still dropped + (proof: test:packages/cli/tests/test_usage_attribution_programs.py) +- [ ] `software-factory`'s subject identity is the release tag together with + the catalog fingerprint, and two builds at one tag with different + catalogs are two versions + (proof: test:packages/cli/tests/test_inventory_census.py) +- [ ] A profile in `local-only` spools and never opens a connection, proven the + same way `off` is + (proof: test:packages/cli/tests/test_usage_upload.py) +- [ ] Every feedback report from this corpus names the exact version it is + about and carries the first-party authorship disclosure + (proof: test:packages/cli/tests/test_cli_courses_source_link.py) +- [ ] One report per family exists, written after a real task on a real + repository + (proof: unspecified:a report an operator writes is not producible by a + test; what a check can hold is that the record exists and names a version, + which the criterion above already does) + +**Exit condition:** on both machines, every artifact loaded into a session +resolves to a named version; `logion usage pending` is non-empty after an +ordinary working day with no change made to amy or to software-factory; the +stale companion skill, the two drifted `factory-*` skills and the orphaned +`amy-develop` are each reported by name rather than discovered by hand; and at +least one feedback report exists per family, naming the exact version it is +about. diff --git a/plans/phase-16.2-typed-evaluators-and-skill-reference-evaluator.md b/plans/phase-16.2-typed-evaluators-and-skill-reference-evaluator.md index 5dba0368..1f45db9c 100644 --- a/plans/phase-16.2-typed-evaluators-and-skill-reference-evaluator.md +++ b/plans/phase-16.2-typed-evaluators-and-skill-reference-evaluator.md @@ -100,7 +100,7 @@ Descriptor binds evaluator ID/version/source digest, supported resource/media ty - Resolve an exact `agent_skill` version and verify bundle digest before extraction. - Create project-scoped install inside the job workspace using existing CLI install helpers as a library; do not shell out to `logion skills install` when a safe library call exists. - Project contents: fixture repo, control/assisted workspaces, pinned skill, harness projection, and output only. No global `~/.logion` or agent skill directory writes. -- Run control and assisted arms with the same agent/provider/config/budgets. Record any unsupported seed determinism as a limitation. +- Run control and assisted arms with the same agent/provider/config/budgets, the same harness stack, and the same iteration budget. An arm wrapped in an orchestration loop the other arm did not get is not a control; see [16.1's environment fields](phase-16.1-eval-contract-and-reference-runner.md#a-pair-is-not-always-a-pair). Record any unsupported seed determinism as a limitation. - Capture declared vs observed capabilities, tool calls, assertion vector, tokens/cost/latency, and cleanup result. - Skill instructions are untrusted input and cannot modify evaluator policy/assertions. @@ -133,13 +133,13 @@ Descriptor binds evaluator ID/version/source digest, supported resource/media ty - Evaluator plugin interface keyed by resource type and media type. - Skill reference evaluator with project-scoped install, baseline/control, assisted run, capability observation, and cleanup. - Typed metric namespaces and explicit units/directionality. -- Evaluator identity, version, source digest, and environment fingerprint in every result. +- Evaluator identity, version, source digest, and the closed 16.1 environment fields — full harness stack, model, iteration budget — in every result. - Quarantine for evaluator/resource incompatibility. ## Mandatory proving-ground scenario Use [the common gate](agent-proving-ground-phase-gate.md) and add -`builtin:phase_16_2_skill_evaluator`. +`builtin:skill_reference_evaluator`. - **Prompt:** “Evaluate whether this indexed debugging skill improves completion of the supplied repository task. Compare control and assisted arms, report @@ -172,3 +172,6 @@ what the markers below leave unproven. - [ ] The project-scope install is removed after the run and a canary proves no global installation changed. (proof: unspecified:no assertion proves the project-scope install is removed after the run) +- [ ] Control and assisted arms are refused when their harness stack or + iteration budget differs, not merely reported as a limitation. + (proof: unspecified:api.eval_arms_comparable retains budgets and inputs and does not retain the harness stack) diff --git a/plans/phase-16.4-deterministic-replication-and-agreement.md b/plans/phase-16.4-deterministic-replication-and-agreement.md index 1a4a5f42..5991146c 100644 --- a/plans/phase-16.4-deterministic-replication-and-agreement.md +++ b/plans/phase-16.4-deterministic-replication-and-agreement.md @@ -49,7 +49,7 @@ isolated runner executions reconcile through the implemented code. ## Replication policy object -Versioned policy fields: eval contract digest, eligible evaluator digests, minimum total results, minimum distinct operators/independence groups, permitted environment equivalence class, deadline, deterministic fields, tolerated numeric precision, and decision rule. Policy digest is stored in every reconciliation. +Versioned policy fields: eval contract digest, eligible evaluator digests, minimum total results, minimum distinct operators/independence groups, permitted environment equivalence class (which never spans differing harness stacks or iteration budgets), deadline, deterministic fields, tolerated numeric precision, and decision rule. Policy digest is stored in every reconciliation. ## Canonical result digest @@ -116,3 +116,4 @@ Use [the common gate](agent-proving-ground-phase-gate.md) and add - Policy changes create new decisions instead of rewriting old ones. - Reconciliation result is reproducible offline from its bundle with the API unavailable. - Same binary run twice under two runner IDs owned by Logion does not satisfy the independent threshold. +- Two results whose harness stacks or iteration budgets differ are never counted as replications of each other, however close their numbers. diff --git a/plans/phase-16.5-eval-attestations-and-cross-node-authority.md b/plans/phase-16.5-eval-attestations-and-cross-node-authority.md index 68978372..7e8e275c 100644 --- a/plans/phase-16.5-eval-attestations-and-cross-node-authority.md +++ b/plans/phase-16.5-eval-attestations-and-cross-node-authority.md @@ -49,7 +49,7 @@ the backend result. No actual resource use means no feedback. ## Attestation predicate and cryptography -- Add `aktp.eval.result/v1` to the public protocol package. It embeds/references eval contract digest, subject digest, evaluator descriptor digest, runner receipt digest, normalized result digest, environment class, artifacts, outcome, limitations, and replication group/decision when present. +- Add `aktp.eval.result/v1` to the public protocol package. It embeds/references eval contract digest, subject digest, evaluator descriptor digest, runner receipt digest, normalized result digest, environment class — including the full harness stack and iteration budget of [16.1](phase-16.1-eval-contract-and-reference-runner.md#a-pair-is-not-always-a-pair) — artifacts, outcome, limitations, and replication group/decision when present. - Reuse the 15.11 canonicalization, signature, issuer, key rotation, and artifact code. Do not create an “eval signature” subsystem. - An attestation is cryptographically `valid|invalid|unverifiable`; local authority is separately `accepted|rejected|insufficient|expired`. API/UI must expose both. @@ -110,3 +110,4 @@ Use [the common gate](agent-proving-ground-phase-gate.md) and add - No single global score or hidden issuer weighting is introduced. - Public verifier reaches the same explanation tree as the backend for every golden policy/evidence bundle. - An invalid signature can never be rescued by a permissive authority policy. +- A local policy can require a declared harness stack and iteration budget, and an attestation omitting either resolves to `insufficient` rather than `accepted`. diff --git a/plans/phase-16.9-benchmark-field-reconciliation.md b/plans/phase-16.9-benchmark-field-reconciliation.md index 5609deae..e1410d17 100644 --- a/plans/phase-16.9-benchmark-field-reconciliation.md +++ b/plans/phase-16.9-benchmark-field-reconciliation.md @@ -88,7 +88,7 @@ Shadow-mode alerts on Logion dogfood for four weeks. Human triage labels useful/ ## Build -- Join compatible signals by resource digest, metric definition, harness class, and time window. +- Join compatible signals by resource digest, metric definition, harness stack, iteration budget, and time window. - Drift and divergence detection with minimum sample and confidence metadata. - Investigation artifacts linking representative receipts, evals, and environment differences. - Optional bounty draft; never auto-publish or auto-pay from telemetry alone. @@ -122,3 +122,5 @@ Use [the common gate](agent-proving-ground-phase-gate.md) and add - No automatic ranking penalty or payout decision consumes reconciliation until a later explicit policy phase. - Declared ASM/provider claims, receipts, Logion observations, and reconciliation outputs remain separately attributable and digest-addressed. +- A benchmark result is never reconciled against a field cohort whose harness + stack or iteration budget differs from the one the benchmark ran under. diff --git a/plans/positioning-and-independence.md b/plans/positioning-and-independence.md index c5f391cb..6fb2842d 100644 --- a/plans/positioning-and-independence.md +++ b/plans/positioning-and-independence.md @@ -10,9 +10,9 @@ before positioning arguments, and before any decision that moves commerce earlier in the sequence. Exit condition: the language-discipline rules below are wired into public-copy -review (no "the network validates" before issuer #2, no pulled-forward -commerce), and the issuer-#2 milestone entry records its first real entry when -one exists. +review (no "the network validates" before issuer #2, no "Google of tools", no +pulled-forward commerce), and the issuer-#2 milestone entry records its first +real entry when one exists. ## Naming: stop calling it a marketplace @@ -22,12 +22,18 @@ framing is what produced abundant supply with no demand. On the critical path the product is: ```text -local skill-hygiene tool → measurement service → publisher observability +published measurement → the CLI that produces field evidence → publisher observability + (acquisition) (retention) (monetisation) ``` -Each stage works alone, in order, with one person. The shipped commerce rails -(credits, entitlements, Stripe Connect, bounties, ledger) remain built and -remain the *later* monetisation of a trust layer. They are not the wedge. +Each stage works alone, in order, with one person, and the first one works with +**zero** — a controlled evaluation needs nobody's permission and runs at N=1. +The shipped commerce rails (credits, entitlements, Stripe Connect, bounties, +ledger) remain built and remain the *later* monetisation of a trust layer. They +are not the wedge. + +> Local skill hygiene used to sit in the first box. It is retention, not +> acquisition — see [The wedge](#the-wedge). The monetisation sentence: @@ -37,33 +43,98 @@ Bounty revenue is remediation revenue, not sales commission. The incentive is aligned with the measurement being honest, which is the opposite of a store rating its own inventory. -## The end-user wedge +## The wedge + +> **Corrected 2026-09-01.** This section used to say the wedge was local mode — +> *"which of my installed skills did I actually use?"* — and called it "the +> end-user wedge". [`public-answer-surface.md`](public-answer-surface.md), written +> ten days later, already contradicted it: **"the public box is acquisition; the +> CLI is retention."** The newer file was right and this one was binding on public +> copy, so the stale version was shaping the pitch. What follows replaces it. + +### Acquisition is a rendered finding, not an install + +The wedge is **one published measurement of a popular artifact, readable with no +account and no install.** Everything else — the CLI, local mode, consent, the +cohort — is what happens after someone already believes the number. + +```text +acquisition a rendered finding on a public URL (no install, no account) +retention "which of mine do I actually use?" (the CLI) +monetisation cohort evidence a publisher cannot get (the publisher) +``` + +The order matters because the middle rung cannot carry the first. That was the +error being corrected. + +### The individual local-mode pain is real, and its carrier is an org + +The pain is not imaginary: `anthropics/claude-code#35319` — 183 skills in one +org, up from 67 in four weeks, no visibility, context budget unjustifiable — +filed and closed with no maintainer response. The mistake was assigning it to an +*individual* as a reason to install. + +Field check, 2026-08-31, an informal thread of six working senior engineers +(n=6, self-selected, not a study — recorded because it is the only direct +evidence available and it points one way): + +- three report using **zero** skills, one of them after trying roughly a hundred; +- one uses only skills he wrote himself and does not revisit them; +- none reported an inventory-hygiene problem; +- none could state what Logion does after reading the landing page. + +An individual with four skills does not have a hygiene problem. **The 183-skill +inventory belongs to an organisation**, and an org is reached through a person +who already trusts the measurement. So local mode is retention and expansion +inside an account, not the thing that opens one. + +`local-only` consent is still not a privacy concession — it is the default +product, nothing leaves the machine, and `logion usage pending` shows exactly +what exists locally. That property sells the CLI to someone already convinced. +It does not convince anyone. + +### The buyer is the person who already gave up + +The same thread produced the strongest demand signal in the file, from its most +hostile participant: *"of about a hundred skills I tested, not one was actually +useful."* That is the workspace's unverified *~67% of skills fail in practice* +restated harder, first-hand, by someone with no stake in it. -Publisher-side value — evidence about your own artifact across installations you -do not control — is settled. The unsolved side is why an individual installs -Logion at all, and without that side the publisher pitch has no cohort behind it. +It reframes the question the product answers. The copy currently addresses an +enthusiast choosing among 300 candidates. The reachable audience is the engineer +who **already stopped**, and their question is not *"which one is best?"* but +*"is it worth trying again?"* Those are different first screens: a ranking +answers the first, a finding answers the second. -The answer is not "join a marketplace". It is that local mode is selfishly -useful with zero network participation: +Corollary for [Step 0 selection](../maintainer documentation: measurement-publication-playbook.md): +a subject that provokes public *disagreement* outranks one that is merely +popular. Observed on `caveman` in the same thread — *"it is not consensus that +caveman is good; a lot of people dislike its output"* — which fires trigger 1 +(a falsifiable published number), trigger 2 (demand) and trigger 3 (reach) at +once. That is why it stays the preferred first subject in +[`next-steps.md`](next-steps.md#the-shape-of-step-6). -- *Which of my installed skills did I actually use? Which never fire and only - cost context? Which are duplicates or stale?* This is the pain in - `anthropics/claude-code#35319` — 183 skills in one org, up from 67 in four - weeks, no visibility, context budget unjustifiable — filed and closed with no - maintainer response. -- `local-only` consent is not a privacy concession. **It is the default - product.** Nothing leaves the machine; `logion usage pending` shows exactly - what exists locally. +### Reproduction is the substitute for reputation -This is the Snyk/Dependabot pattern: the user installs for their own benefit and -ecosystem data is the by-product. Nobody installs Dependabot to help GitHub map -vulnerabilities. +The objection that has no answer in this file yet, stated plainly in the same +thread: *"if you were somebody, people would believe the number."* It is +correct, and independence does not answer it — independence is about conflict of +interest, notoriety is about whether anyone reads you at all. -Network participation is a **separate, later step** whose incentive is -comparison: your own numbers mean nothing without a cohort. Same trade offered -to the publisher. +The only asset an unknown issuer has is that its result can be **re-run by +someone else**. That converts reputation from something Logion must accumulate +into something a stranger can verify in one command. But it is only real when +somebody actually does it, so it has to be an exit condition rather than an +affordance: -Two objections, each answerable in one sentence: +> **One external reproduction of the first public measurement is worth more than +> any install count**, and it is the cheapest partial down-payment on +> [issuer #2](#issuer-2--the-milestone-that-makes-the-thesis-true) available +> before federation exists. + +Tracked as part of [step 6](next-steps.md#execution-order). + +### The two standing objections - *"It will eat my tokens."* The observation path is a subprocess hook, not a model call — zero tokens. Anything that does cost tokens stays separable. @@ -71,10 +142,26 @@ Two objections, each answerable in one sentence: prompts, paths, arguments and content **by schema, not by policy**; the user can read, export and delete it; `DO_NOT_TRACK` forces `off`. -**Distribution consequence:** the publisher is the channel for end-user installs, -not the reverse. A publisher who wants cohort data asks their own community to -install the companion. Logion supplies the instrument; the publisher supplies -the audience. +### Distribution consequence + +The publisher is the channel for end-user installs, not the reverse. A publisher +who wants cohort data asks their own community to install the companion. Logion +supplies the instrument; the publisher supplies the audience. This is still the +Snyk/Dependabot shape — the user installs for their own benefit and ecosystem +data is the by-product — but the *first contact* is the report, not the tool. + +### Naming discipline that follows from all of this + +"Google of tools" tested badly and must not be used. In the 2026-08-31 thread it +produced, in order: *"isn't that just a README"*, *"a hub of skills?"*, *"a +marketplace of markdown"*, and *"HuggingFace for skills"* — every one of them the +marketplace category this document exists to leave. Google is an *index*, and an +index is the commodity half of the product. + +Lead with the verb instead: **nobody proves that a skill works; Logion does.** +In the same thread that sentence is what moved the hardest sceptic from "what is +the appeal" to "what is the methodology", which is the question the product is +built to answer. ## Why Logion is not Apify @@ -137,6 +224,41 @@ Two practical consequences: `future-roadmap/smart-payments-and-metered-capabilities.md` and argues against binding the model to one-time course purchase. +## What the hubs actually publish, checked 2026-08-27 + +The differentiation argument should rest on what the pages render, not on a +characterisation of them. + +| Hub | What a skill page shows | Behavioural evidence | +| --- | --- | --- | +| skills.sh | installs, GitHub stars, first-seen date, three security audits (Gen Agent Trust Hub, Socket, Snyk) | **none** — no test results, no reviews, no ratings, no version history | +| ClawHub | downloads, installs, stars, lineage, ownership, docs, package integrity, audits | none found | +| LobeHub | catalogue and discovery | none found | + +Every one of them measures **provenance and popularity**. None measures +**behaviour**. That is the whole gap, and it restates the standing rule: install +counts are commodity; only use and outcome differentiate. + +### The self-demonstrating case + +`skills.sh/vercel-labs/skills/find-skills` — 3.1M installs, 29.7K stars — states +in its own description that the quality criteria it applies when recommending +other skills are **installs above 1K, source reputation, and GitHub star count**. + +The most-installed skill whose job is finding good skills ranks by popularity, +because popularity is the only signal the ecosystem publishes. The argument for +Logion does not need to be made in prose; it is already written by the ecosystem +into its most-installed artifact. + +The same page carries **Snyk `Warn` alongside two passing audits**, and nothing +reconciles them — because a security audit is not evidence of function, and the +market has nowhere to put that difference. + +Note also that `find-skills` exists on both skills.sh and ClawHub under the same +name. Neither hub can tell a user what the other holds. That is step 2 of +[`next-steps.md`](next-steps.md) appearing unprompted in the first example +anyone picks. + ## The garden without walls Apify measures what it hosts. Logion measures what exists — including artifacts @@ -259,6 +381,17 @@ Language discipline, binding on public copy: validate itself is, by the workspace's own coalition-wealth test, indistinguishable from a collusion ring. Cf. `future-roadmap/economic-network-and-rewards.md`. +- **Never describe Logion with an index metaphor** — "Google of tools", "hub", + "catalog of skills". Tested 2026-08-31 and it routes every listener straight + back into the marketplace category. Lead with the verb: *nobody proves that a + skill works; Logion does.* +- **A measurement is never an endorsement, and it must say so where it is + rendered.** A published finding about an artifact is scoped to one contract, + one environment, one subject version. It is not a safety certification, not a + compliance attestation, and not a recommendation to install. That disclaimer + currently lives only in internal docs; it belongs next to the finding, because + the first artifact Logion measures that later turns out to be malicious will + be read as *"Logion approved it"* unless the page already said otherwise. ## Issuer #2 — the milestone that makes the thesis true @@ -303,3 +436,37 @@ consumer policy under which it verified. - **skills.sh** already ships install telemetry, weekly installs per skill, per-agent-platform breakdown, and all-time/24h leaderboards. **Install counts are commodity.** Only use and outcome differentiate. +- **Warp** is the closest thing to a direct competitor that exists, and it is + worth reading precisely rather than dismissing. Warp Factories orchestrates + **third-party harnesses by name** — `claude code`, `codex`, `cursor`, "any + MCP-capable coding agent" — and advertises evals, custom scoring for spec + adherence and defects caused, ROI metrics, an Agent Kits gallery, benchmark + comparisons **across models and harnesses**, and a self-improvement loop where + "agents study patterns across runs and open PRs against your factory config". + That is Braintrust plus Apify plus an improvement loop, and it disproves the + earlier claim that nobody ranks model-harness pairs. + **The boundary is the subject, and their own page draws it:** those benchmarks + measure *"your factory's agents executing your specific workflows, **not + third-party artifact quality**"*. Four consequences follow — the subject is + your agents rather than someone else's artifact; results stay private to the + customer instead of being publicly addressable by an agent about to install + something; it requires adopting Warp; and benchmarking harnesses while selling + the orchestration platform next to an artifact gallery is the house rating + inventory adjacent to its own store, which is the structure that produced + `audit-tools.ai` inside Apify. Asked publicly what outcome cost telemetry + changes, the founder's answer was to pivot to "benchmark models on your own + code" — deeper into the first-party perimeter, not out of it. + **Read them as the best available candidate for issuer #2**, not as a threat: + they already produce per-run, per-model, per-harness, per-cost evidence with + pass rates, and AKTP is the envelope an operator would emit it through. +- **Braintrust** ($80M Series B, Feb 2026) is not a competitor but is the + category a reader defaults to, which makes it a positioning problem rather + than a competitive one. It is eval and observability for the LLM app *you* + wrote — your dataset, your task function, your scorers, your traces. Same + structural boundary already stated for trajectory.ai, now with a reference the + market recognises. Use it, do not invent vocabulary: **Braintrust evaluates + the agent you wrote; Logion evaluates the parts you installed inside it — the + ones you did not write.** A business-side reader who follows the space + understood ~30% of the current landing precisely because he assigned Logion to + the Braintrust category and then found nothing on the page answering that + category's questions. diff --git a/plans/public-answer-surface.md b/plans/public-answer-surface.md new file mode 100644 index 00000000..fd6a1bd1 --- /dev/null +++ b/plans/public-answer-surface.md @@ -0,0 +1,237 @@ + + +# The public answer surface + +> **Why this file exists:** as of 2026-08-27 no plan describes a way to get an +> answer out of Logion without installing the CLI. The only public web surfaces +> planned anywhere are the landing and one evidence report page +> ([`17.6`](phase-17.6-public-narrative-and-landing-truth-pass.md), +> [`release-0.2.md`](release-0.2.md)). Loop B's answer is already designed to +> need nobody — this is the surface that exposes it. +> **Honesty boundary:** this surface answers with evidence that already exists. +> It never issues a new verdict about a third party on demand. + +## The split it enforces + +The axis is not *site versus CLI*. It is **consuming the answer versus producing +the evidence**. + +| | Consuming the answer | Producing field evidence | +| --- | --- | --- | +| Surface | public URL, no account | the CLI | +| Install required | none | yes | +| Question it answers | "does this artifact work?" | "which of *my* installed artifacts do I actually use?" | +| Analogue | Google | Analytics / Dependabot | + +`release-0.2.md` already decided that Loop B's launch answer is **controlled +evaluation, which "needs nobody" and works at N=1**. That means the answer never +depended on the asker having installed anything. The product is ready to be +zero-install; only the surface is missing. + +The two surfaces do not compete. The public box is acquisition; the CLI is +retention. + +## The category problem this fixes first + +A business-side reader who follows MCP/eval/RAG discourse read `logion.sh` on +2026-08-26 and reported understanding roughly 30% of it, seeing neither the +problem nor the mechanism. The diagnostic detail is which product he reached for +to fill the empty category: **Braintrust** — eval and observability for a team +shipping *its own* LLM app (its unit is your dataset, your task function, your +scorers). + +Once a reader assigns that category, nothing else on the page can land. A +Braintrust-shaped reader needs no index, no cross-hub reconciliation, no issuer, +no attestation. He read 30% because the other 70% answers questions his assumed +category does not ask. + +An empty category is occupied by borrowing the full one next to it — the same +move [`positioning-and-independence.md`](positioning-and-independence.md) +already prescribes for Apify vocabulary, applied to the category rather than the +metric: + +> **Braintrust evaluates the agent you wrote. Logion evaluates the parts you +> installed inside it — the ones you did not write.** + +The rendered output disambiguates the category better than any tagline: when the +subject on screen is a third party's artifact, nobody confuses it with +first-party observability. + +## The surface + +```text + L O G I O N + + ┌──────────────────────────────────────────────┐ + │ paste a skill, plugin, MCP server or model │ + └──────────────────────────────────────────────┘ + + [ Does it work? ] + + ──────────────── recent findings ──────────────── + (results render immediately below, on first load) +``` + +The box stays empty, as Google's does. What makes an empty box legible is that +the output shape is visible below it without asking — Google can omit that +because everyone already knows what a search result looks like, and nobody knows +what a Logion result looks like. + +### Rule 1 — the scroll renders findings, not inventory + +This is the difference between the product and a marketplace listing, and it is +a rendering decision, not a copy decision. + +```text +INVENTORY (rebuilds the marketplace) FINDING (renders the product) +┌──────────────┐ ┌──────────────┐ ┌────────────────────────────────┐ +│ pdf-tools │ │ web-scraper │ │ pdf-tools v2.1.0 │ +│ ★ 4.8 12k │ │ ★ 4.6 8k │ │ fails 40% of extract-table │ +└──────────────┘ └──────────────┘ │ 200 runs · method published │ + └────────────────────────────────┘ +``` + +A grid of artifacts with names and counters is the landing +`../maintainer documentation: landing-page.md` is being rewritten to stop being. The object on +screen must be the verdict, not the item. + +### Rule 2 — the box never returns an empty screen + +An honest system will answer "nobody has measured this yet" for most of the +index at launch. If a visitor's first three attempts return nothing, the box +teaches that the answer is nothing, and it becomes a liability. + +So the layered answer `release-0.2.md` already defines becomes the box's +response ladder, and the **free layer has to cover the tail**: cross-hub +presence and version coverage, install counts from the hubs that publish them, +scanner results, permissions, license, provenance, freshness, and edges between +artifacts. That layer costs zero inference and spans the whole index, which is +what makes the box survivable before the evaluated head is large. + +`"no measurement yet"` is a valid terminal answer, but it always carries a +request-a-measurement action, and those requests are the demand queue that +selects what to evaluate next. + +### Rule 2.1 — write for the person who already gave up + +The response ladder decides *what* is shown. This decides *who it is written +for*, and it is the correction recorded in +[`positioning-and-independence.md`](positioning-and-independence.md#the-buyer-is-the-person-who-already-gave-up). + +The reachable reader is not an enthusiast comparing 300 candidates. It is an +engineer who tried a pile of skills, concluded none of them worked, and stopped +— *"of about a hundred skills I tested, not one was actually useful"* (field +thread, 2026-08-31). Their question is **"is it worth trying again?"**, not +"which is best?". + +Consequences for this surface: + +- A ranked list answers the enthusiast's question and bounces this reader. A + single rendered verdict answers theirs. This is why Rule 1 is a rendering + decision and not a taste preference. +- A negative finding is not a downer, it is the credential. It is the proof that + the page will tell them the truth next time, which is the only thing that + restores a burned reader. +- Do not open with ecosystem enthusiasm. This reader has already priced in that + most of it does not work; agreeing with them is the fastest way to be + believed. + +### Rule 3 — the CTA after a result is a different question, never a paywall + +"Install the CLI to see more" puts a paywall on the answer and destroys the +property the box just bought. The install prompt offers something the web cannot +answer: + +> You looked up one skill. Which of the ones you already have installed do you +> never actually use? + +That is the pain in `anthropics/claude-code#35319` and the local-mode wedge +already stated in `positioning-and-independence.md`. The public answer stays +free permanently. + +### Rule 4 — one URL, two readers + +Every indexed subject gets a stable URL that content-negotiates: HTML for a +browser, JSON/Markdown for an agent. `packages/landing/` already does this for +`landing.md`, `llms.txt`, `llms-full.txt` and `/design.txt`; the result page +inherits the same mechanism, plus an MCP endpoint. + +This is the distribution answer to "no harness will bundle Logion." No harness +bundled Google either — it was a URL you could paste. An agent that can check an +artifact *before* installing it is the actual buyer, and it reaches this surface +with nobody installing anything. + +## What this collapses + +The landing rewrite ([`17.6`](phase-17.6-public-narrative-and-landing-truth-pass.md)) +and the "one evidence report page" required by `release-0.2.md` are the **same +deliverable**, and the order inverts: the report page is built first and the +landing is it. This reduces 0.2 scope rather than adding to it. + +Do not build a dashboard or a charting framework. One rendered result, done +well, plus the version-over-version chart already required. + +### Rule 5 — every finding renders its own scope boundary + +Not in a footer, not in a terms page: **next to the result, on the same screen.** + +```text +┌────────────────────────────────────────────────────────────┐ +│ pdf-tools v2.1.0 │ +│ fails 40% of extract-table · 200 runs · method published │ +│ │ +│ measured under contract , harness X vY, model Z vW │ +│ not a safety review · not a certification · not advice │ +│ to install │ +└────────────────────────────────────────────────────────────┘ +``` + +Two reasons, and the second is the one that has no owner anywhere else in the +plans: + +1. The layered answer is worthless if the reader cannot tell which layer they + are looking at, and a limit rendered elsewhere is a limit nobody reads. +2. **Publishing a number about an artifact is read as blessing the artifact.** + The moment something Logion measured turns out to be malicious — and at + ecosystem scale that is when, not if — the default public reading is + *"Logion approved it"*. The only defence that works is the one that was + already on the page before the incident, in the same visual weight as the + verdict. `measurement-publication-playbook.md` already says a measurement is + *"not a ranking, not a leaderboard, not a safety certification, not a + compliance attestation, and not a recommendation to buy"* — that sentence is + currently internal-facing only, and this rule is what makes it public. + +Scanner results may appear as static evidence with their own issuer named. They +never graduate into a safety verdict, because a security audit is not evidence +of function and Logion does not run the artifact on the reader's machine. + +## What this surface must not do + +- **It never issues a new verdict on demand.** An arbitrary paste returns + existing evidence; it does not trigger an evaluation whose result is published + about a named third party without the author contact step in + [`../maintainer documentation: measurement-publication-playbook.md`](../maintainer documentation: measurement-publication-playbook.md). + The findings in the scroll are deliberately published measurements that + already cleared that process. +- It never blends layers into one opaque score, and never renders a badge. +- It never claims network validation while Logion is the only issuer. +- **It never implies a measured artifact is safe, endorsed, or recommended.** + See Rule 5. + +## Preconditions + +**Exit condition:** one public URL answers "does this artifact work?" with no +account and no install, renders at least one real retained finding on first +load, and content-negotiates a machine-readable answer for an agent. When +that URL is live and the retained finding it renders is one produced by a +real sealed run, this plan is finished — not at the merge of any PR. + +| Needs | From | +| --- | --- | +| Free layer covering the tail | step 2, cross-hub install reconciliation | +| At least one rendered finding | step 5 (`16.1`/`16.2`) — one measured artifact | +| Subject types beyond skills | `model` already exists; `eval_contract` added by [`16.1`](phase-16.1-eval-contract-and-reference-runner.md) | + +Models and eval contracts matter here for a specific reason: they carry rich +public metadata on day one and skills do not, so adding them is the fastest path +to Rule 2 holding. diff --git a/plans/release-0.2.md b/plans/release-0.2.md index faca20f7..3992d266 100644 --- a/plans/release-0.2.md +++ b/plans/release-0.2.md @@ -94,7 +94,10 @@ query, the first experience is worthless. So the answer to Loop B is layered, and the layers must be visibly labelled: 1. **Controlled evaluation** (16.2) — Logion's own measurement. Available on - day one, needs nobody. This is the primary answer at launch. + day one, needs nobody. This is the primary answer at launch. It measures + **isolated task performance**, which is a weaker claim than sustained use + across a session, and the layer label must say so — see + [The sustained-use boundary](#the-sustained-use-boundary). 2. **Static evidence** — scanner results, capability manifest, permissions, license, freshness, source provenance. 3. **Field cohort** — deterministic receipts and agent feedback, shown only @@ -122,6 +125,16 @@ publisher losing interest. Indexing them is the cheapest way to stop Loop B saying "nothing" on day one, and it costs no inference and needs nobody's permission. +**The cheapest first specimen is a published benchmark about a `model`.** A +benchmark that names a model version and the harness it ran in already carries +every field this layer requires — issuer, date, a determinable subject revision, +a stated method, a public URL — and it exercises `superseded` naturally, because +models ship new versions faster than skills do. +[arXiv:2603.13428](https://arxiv.org/abs/2603.13428) is one such specimen: it +reports a named model in a named harness at 38.03%. Proving the layer against a +benchmark is a cheaper first test of the same storage class than hunting a blog +post about a skill, and `model` is already in scope for the answer surface. + It is also the layer that has no storage class today. The capability-profile ladder is `observed` / `candidate` / `attested`, each with exactly one producer, and an unsigned external study is none of the three: Logion's @@ -136,6 +149,38 @@ without Logion in the loop.** A citation Logion has to transcribe is an index; a predicate an outside issuer can sign about their own work is the first step to issuer #2, which is the actual thesis milestone. +#### The sustained-use boundary + +Layer 1 is the launch answer, and a controlled evaluation is an isolated-task +measurement by construction — that is exactly what makes it reproducible and +what lets another runner replicate it. It is not evidence about behaviour +sustained across a working session, and 0.2 must not let the rendering imply +otherwise. + +Two independent published results bound this, from different groups, months +apart: + +- [arXiv:2603.13428](https://arxiv.org/abs/2603.13428) holds agent, model, + repository and verification constant and varies only whether tasks are + independent or dependency-chained: top scores fall from 80–90% on isolated + tasks to a maximum of 38.03% under continuous evolution, with a milestone + resolve rate of 13.37%, and roughly 57% of root causes attributed to failures + inherited from earlier steps rather than generated locally. +- [arXiv:2608.27454](https://arxiv.org/abs/2608.27454) states the same gap from + the other side, declaring very long-horizon tasks — hundreds of environment + actions, multiple hours — outside its evaluated scope. + +The consequence for this release is a blind spot, not a feature: the +deterministic receipt is a per-invocation record, so a failure carried in from +an earlier step is invisible to the unit Logion records. Say that, in the layer +label and in the disclaimer. Under +[`../maintainer documentation: measurement-publication-playbook.md`](../maintainer documentation: measurement-publication-playbook.md) +a stated blind spot is also a demand signal for what to measure next. + +Do not try to close it in 0.2. The archetype that would measure across a +sequence is deferred — see +[What 0.2 explicitly does not include](#what-02-explicitly-does-not-include). + ### Loop C — claim, sell, improve (NOT 0.2 — target 0.3) ```text @@ -213,12 +258,19 @@ So the honest public statement is: *for a remote MCP server, Logion can report that an endpoint was used and how the agent judged it, but cannot bind that to an exact immutable version, and reports only what the client saw.* Server-side truth requires the operator to instrument -([`15.11.1`](phase-15.11.1-publisher-integrated-consented-observation.md)) — +(designed, not built — see +[`normative-carry-overs.md`](normative-carry-overs.md#publisher-integrated-observation--designed-not-built)) — which is the upsell, not a gap to paper over. Remote execution never authorises TLS interception, credential access, probing, load testing, or provider-side modification. +**This table bounds what can be attributed, not what a measurement means.** The +second boundary is [The sustained-use boundary](#the-sustained-use-boundary) +under Loop B, and the public disclaimer needs both: one says which subjects +Logion can bind to an exact version, the other says what a passing evaluation +of that version does and does not claim. + ## Public surface required for 0.2 The agent-first documentation stance was correct when the user was an agent @@ -227,21 +279,32 @@ must be convinced is a human at a company deciding whether to trust Logion with usage data and whether to fund a bounty. That person opens a browser, spends thirty seconds, and leaves. -Required: +Required, and they are **one** deliverable, not four — +[`public-answer-surface.md`](public-answer-surface.md) owns the whole of it: -1. **Landing truth pass** — [`17.6`](phase-17.6-public-narrative-and-landing-truth-pass.md), +1. **A public answer with no account and no install.** One URL where pasting a + skill, plugin, MCP server or model returns what is known about it. This is + the surface Loop B was always designed for: its launch answer is controlled + evaluation, which "needs nobody" and works at N=1, so the answer never + depended on the asker having installed the CLI. Until now no plan described + a way to get an answer out of Logion without installing it. +2. **Landing truth pass** — [`17.6`](phase-17.6-public-narrative-and-landing-truth-pass.md), pulled forward out of Phase 17. The current landing describes a marketplace and shipping 0.2 against it publishes a false description of the product. -2. **A human-readable explanation of the loop** — measure → evidence → - improvement → re-measure. + It is executed *through* item 1: the landing **is** the answer surface. 3. **One evidence report page.** A permanent URL rendering one measurement well: the chart, the method, the eval-contract digest, the subject version, the limits, `n` where field data appears, and the command to reproduce. yukon.org's benchmark-evolution view is the reference for what "showing the - number" looks like when evidence is the product. + number" looks like when evidence is the product. Built **first**, because the + landing renders it rather than linking to it. 4. **A version-over-version chart.** The single most valuable visual is the same subject measured across versions — it makes improvement visible, which - is the entire thesis in one image. + is the entire thesis in one image. It must refuse to plot two points whose + harness/model pair differs without saying so on the chart — otherwise a + harness upgrade renders as an artifact improvement. +5. **A human-readable explanation of the loop** — measure → evidence → + improvement → re-measure — placed *below* a rendered result, never above it. Scope discipline: build one page that renders one report beautifully. Generalise into a dashboard only once there are enough reports to justify it. When the @@ -249,22 +312,120 @@ product is evidence the visualization is the deliverable — that is an argument for making one excellent, not for building a charting framework around a single data point. +**Do not put ecosystem statistics in the hero.** Aggregate problem statistics are +what a page shows when it cannot show the output. A rendered finding — "this +skill fails 40% of task class X" — is a stronger argument than any market-level +number, and the stats belong in the report body, the call, and the launch post. + +**Category discipline.** A business-side reader who follows the space read the +current landing and understood ~30% of it, seeing neither the problem nor the +mechanism, because he filled the empty category with Braintrust — first-party +eval and observability for the app *you* wrote. The correction is to borrow the +adjacent full category rather than invent vocabulary: *Braintrust evaluates the +agent you wrote; Logion evaluates the parts you installed inside it.* + +The engineer-side failure mode is different and was measured separately on +2026-08-31: six working senior engineers read the pitch and landed on "isn't +that just a README", "a hub of skills", "a marketplace of markdown". The shared +cause is that **every index metaphor re-imports the marketplace category this +release exists to leave.** Lead with the verb — *nobody proves that a skill +works; Logion does* — and let the rendered finding carry the rest. + +### The first rendered finding is first-party + +Decided 2026-09-02. The subject of the measurement rendered on the public +surface at 0.2 is Logion's own work — the `logion` skill and +`software-factory` — and not a famous skill from a hub. The third-party +measurement stays where it was, at +[step 6](next-steps.md#execution-order), after the release. + +**This resolves a contradiction, it is not a preference.** Three rules in this +workspace currently cannot all hold: + +1. this release's gate requires the public surface to render at least one real + finding **on first load**; +2. [`next-steps.md`](next-steps.md#execution-order) puts the first public + measurement of a third party at step 6, *after* the publish; +3. [the publication playbook](../maintainer documentation: measurement-publication-playbook.md) + forbids publishing about a third party before author contact, a 7-day + response window, a published takedown policy, and the + [legal items](legal-and-entity.md) whose trigger is "before the first + public measurement". + +Rendering a third-party finding at 0.2 either drags step 6 across the release +line or breaks the playbook. A first-party finding satisfies (1) while +leaving (2) and (3) intact, and it is the only subject that does. + +**It is also the only subject that can carry the release's most valuable +visual.** [Item 4 of the public surface](#public-surface-required-for-02) is +the version-over-version chart, and a third-party skill measured once is one +point. `software-factory` has a version history Logion owns, so the same +subject can be measured across its own versions and the chart shows an +artifact improving. That is the entire thesis in one image, and nothing else +available at 0.2 can produce it. + +Three conditions, all binding: + +- **A closed `first_party` field on the result, not a sentence.** 15.16 + already requires honesty labels for first-party / inconclusive / failed on + the public view; this is that field, decided here because 0.2 renders before + 15.16 ships. +- **Publish the unflattering one.** A first finding that says Logion's own + skill scores well is worth nothing as evidence and confirms the reading six + engineers already gave the landing page — a demo. Under + [audience discipline](#public-surface-required-for-02) a negative finding is + the credential. `software-factory` is opinionated enough to plausibly lose + on a task class, which makes it the better of the two subjects. +- **It does not relax the independence rule.** What + [`positioning-and-independence.md`](positioning-and-independence.md#independence-methodological-now-structural-later) + forbids before issuer #2 is *selling what Logion measures*, not measuring + Logion's own work. Measuring itself and saying so is inside the rule; + measuring itself and rendering it unlabelled is not. + +**The task environment is a real repository, and that is the dogfood loop.** +`nicolasmelo1/itl` (public, Python, under active development) is the candidate: +measuring whether a skill helps on invented fixture tasks measures the fixture. +Two constraints follow. The eval contract pins a **commit**, because a repo +still being written is not a stable subject and a result that cannot name the +tree it ran against is not reproducible. And the loop has to close: a finding +about `software-factory` becomes an improvement to `software-factory`, which +is re-measured as a new version and lands as the second point on the chart — +that is [15.16](phase-15.16-first-party-resource-dogfood-loop.md)'s canonical +loop run for real rather than described. `nicolas-portfolio` is out: finished +work generates no ongoing tasks to measure against. + +**Audience discipline.** The reachable reader is the engineer who already tried +a pile of skills, concluded none worked, and stopped. They ask "is it worth +trying again?", not "which one is best?" — so a negative finding is the +credential, not a downer. See +[`public-answer-surface.md`](public-answer-surface.md) Rule 2.1. + ## Gate checklist for cutting 0.2 Ordered. Everything must be true simultaneously. - [ ] **Loop A end to end on a machine that is not the founder's**, with a - harness hook firing **live** — not a replayed payload. The current - `phase-15.11` gate evidence is a replay and does not satisfy this. + harness hook firing **live** — not a replayed payload. Corrected + 2026-09-02: the live-hook half is already done and the replay claim was + misattributed. `files.observation_from_live_hook` passed in the sealed + `phase-15.11` report for `native_use_observation_and_feedback`, with 12 + hook invocations; the caveat on that gate is about + `remote_private_mcp_feedback`, where the agent emits the observation + itself through `usage observe --stdin` because no harness delivers a + payload to a differently-driven session. What is still open here is + only the machine: every run to date is the founder's, against a local + devrig, with seeded fixtures and role API keys. - [ ] The seven "Still open" items in - [`15.11`](phase-15.11-native-use-observation-linked-feedback-and-reviews.md) + `15.11` are closed, including a **pseudonymous identity tier** — `identity_tier` is always `account` today, so feedback still requires an account, which contradicts the local-first pitch on the first screen. -- [ ] **One envelope, not two.** The live `UsageObservation` spool schema and - the richer `cli/_observation.py` envelope cannot both stay normative. - Adopt one, delete the other. -- [ ] `logion.sh` serves `/.well-known/ai-catalog.json` (today: 404). +- [x] **One envelope, not two.** Settled 2026-08-27: `cli/usage/observations.py` + is the single normative envelope and `cli/_observation.py` is deleted. +- [x] `logion.sh` serves `/.well-known/ai-catalog.json`. Verified live + 2026-08-27: 308 to `www.logion.sh`, then 200 with a valid + `specVersion: "1.0"` document. Note the apex redirect — `api.logion.sh` + returns 403 for the same path and is not the serving origin. - [ ] ASM contact (T1) has fired, before the 15.12 design freeze. - [ ] Loop B answers with a labelled layer, and answers "no measurement yet" honestly when that is the truth. @@ -278,13 +439,78 @@ Ordered. Everything must be true simultaneously. (proof: unspecified:the self-export/import test is specified in 15.17, which is sequenced after this release) - [ ] Loop D coverage table is generated from recorded harness fixtures and fails closed on drift. -- [ ] `16.1`/`16.2` produce a normalized result for a third-party skill through - the reference runner on the local node. +- [ ] `16.1`/`16.2` produce a normalized result for a **third-party** skill + through the reference runner on the local node. Run, not published — see + [The first rendered finding is first-party](#the-first-rendered-finding-is-first-party). + The subject stays third-party here for a technical reason and not a + narrative one: only an artifact Logion does not control exercises the + shapes that break a runner in the field — an odd `SKILL.md`, a version + that will not resolve from an external hub, an absent license, no digest. + Measuring only what Logion authored means the runner never meets an + artifact it cannot parse until a stranger's does. `find-skills` is + already pinned and installed by the 15.11 fixture, so this costs almost + nothing. +- [ ] **The finding rendered on the public surface is first-party**, carries a + closed `first_party` field rather than prose, and is a result that is + *not* flattering to Logion. See + [The first rendered finding is first-party](#the-first-rendered-finding-is-first-party). - [ ] Cycle 0 invariants green: additive `/v1`, contract-audit authoritative, supported CLI/API pairs proven. See [`cycle-0-contract-e2e-hardening.md`](cycle-0-contract-e2e-hardening.md) and [`cli-api-compatibility-matrix.md`](cli-api-compatibility-matrix.md). - [ ] Landing truth pass done; report page live; version chart renders. +- [ ] **The public answer surface answers without an account or an install**, + renders at least one real finding on first load, never returns an empty + screen for an indexed subject, and content-negotiates the same URL for a + browser and for an agent. See + [`public-answer-surface.md`](public-answer-surface.md). +- [ ] **Every rendered finding carries its own scope boundary on the same + screen** — contract digest, harness/model pair, **the task class it + measured**, and an explicit "not a safety review, not a certification, not + advice to install, and not a claim about sustained use". A measurement + published without it reads as an endorsement, and the first measured + artifact that turns out to be malicious then reads as Logion's failure. + Task class is on this list because the same artifact under the same model + can be strongly positive on one class of task and strongly negative on + another — measured across benchmarks in + [arXiv:2608.27454](https://arxiv.org/abs/2608.27454) — so a finding + rendered without its task class is the easiest thing a publisher can + correctly refute, at the moment the + [playbook](../maintainer documentation: measurement-publication-playbook.md) says + exposure is highest. At 0.2 this is a **rendering** requirement satisfied + by the contract's own declared subject; the closed field and the queryable + filter are deferred. + See [`public-answer-surface.md`](public-answer-surface.md) Rule 5. +- [ ] **No public copy uses an index metaphor** ("Google of tools", "hub", + "catalog of skills") and none claims network validation. Enforced by a + landing content test, not by review discipline. See + [`positioning-and-independence.md`](positioning-and-independence.md). +- [ ] The observation envelope carries the **harness version**, and a **model + slug from a closed allowlist** where the adapter can declare it honestly. + Adding fields after publish invalidates every installed hook, because the + spool rejects a mismatched `integration_version` — so this is a + before-publish item for schema reasons. It is now also a product + requirement rather than schema hygiene, which matters because that is what + decides whether it survives a scope cut: evolved-skill value measured + across models spans a large gain to a large regression on the *same* + benchmark — [arXiv:2608.27454](https://arxiv.org/abs/2608.27454) reports + one model lifted from 24.3% to 50.5% while, on that same benchmark, + another model was driven from 50.5% down to 18.1%. Field + evidence that does not name the model is therefore not interpretable + evidence, and evidence recorded at 0.2 is permanent. Contract in + [`normative-carry-overs.md`](normative-carry-overs.md#envelope-fields--decided-2026-08-27-unbuilt); + it has no phase owner because 15.11 closed, so it attaches to whichever + phase next touches the envelope. +- [ ] An eval result names its **full harness stack (outermost orchestrator + first, each layer id+version), model id+version, and iteration budget as + closed fields**, and `logion eval compare` fails closed when any of the + three differs between base and candidate. A pair is not enough: an outer + orchestration loop over an unchanged harness and model lifts the same + subject from 44% to 71% at three iterations and 72.67% at ten + ([arXiv:2609.01481](https://arxiv.org/abs/2609.01481)), so a result that + records only the innermost harness attributes the orchestrator's gain to + the artifact. Owner: + [`16.1`](phase-16.1-eval-contract-and-reference-runner.md#a-pair-is-not-always-a-pair). - [ ] Legal items whose trigger is "before step 2 reaches an external publisher" and "before the first public measurement" are done. See [`legal-and-entity.md`](legal-and-entity.md). @@ -330,3 +556,47 @@ Claims and commercial listings (Loop C), independent runners, cross-node attestation authority, bucket-brigade payouts, a dashboard product, GPU-backed model evaluation, and any public statement that "the network validates" while Logion is the only issuer. + +Three measurement-scope items are deliberately deferred rather than absent. +Recorded here so 0.3 does not relitigate them, and so nothing pulls them forward +on the strength of "the cheapest moment to change a versioned schema is now" — +that argument is true of the observation envelope and false of all three of +these. None of them is yet written into the phase named as its owner. + +- **A sequential eval archetype** — a contract whose result is a per-step + assertion vector plus a degradation slope across a dependency-chained + sequence, rather than one vector for one run. Owner: + [`16.2`](phase-16.2-typed-evaluators-and-skill-reference-evaluator.md). It + cannot land in [`16.1`](phase-16.1-eval-contract-and-reference-runner.md), + because a sequential run has compounding variance: "reproduce" would have to + become "reproduce the slope within an interval", which either weakens 16.1's + determinism guarantee or gets cut on the way in. 0.2 discloses the boundary + rather than measuring across it. Note that this is the shape required to + measure a token-reduction claim honestly — the crossover between accumulated + per-turn input overhead and accumulated output saving is a sequence property, + not a single-shot one — so it is owed to the first public measurement even + though it is not owed to this release. +- **Task class as a closed field on the eval contract, and as a filter on + evidence search.** Owners: `16.2` for the field, + [`16.7`](phase-16.7-evidence-search-and-issuer-aware-ranking.md) for the + filter — it is absent from that phase's typed filter list today, so as + specified evidence search cannot answer "does this work for *my* kind of + work". 0.2 needs the label on the screen, not the index, and 0.2 authors a + handful of contracts under one operator, so mapping them to task classes later + is cheap. The field is already first-class on the *field* side — + `resource_usage_receipts.task_class`, and `resource_feedback` is unique on + `(reporter_subject_key, resource_version_id, task_class)`, so one reporter + deliberately holds different judgements of one version across task classes. + The asymmetry is that the eval side carries no equivalent, which is also what + [`16.9`](phase-16.9-benchmark-field-reconciliation.md) needs in order to + compare a benchmark against a matching field population rather than a mixed + one. +- **Source-model provenance on a resource version** — which model produced or + optimised a skill. It predicts outcomes and is not monotonic in the source + model's capability — skills evolved by another model can outperform + self-evolved skills, and a stronger source model does not necessarily produce + better skills ([arXiv:2608.27454](https://arxiv.org/abs/2608.27454)). It stays out of 0.2 + because it is server-side and additive, so it carries none of the envelope's + before-publish urgency, and because almost nothing in the catalogue can + declare it honestly today. When it lands it is an optional closed-enum field + on `ResourceVersion`, never a required one, and never prose.