Compare commits
68 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| be2c577884 | |||
| c0d00ccd18 | |||
| 722d6c76de | |||
| 56f4895881 | |||
| ca249e4986 | |||
| faf605fdf1 | |||
| 7c8644dc45 | |||
| 195292156f | |||
| abe1b52002 | |||
| e365b24339 | |||
| 5e3e88d26d | |||
| ce6907bd73 | |||
| e2b2f51364 | |||
| 0f2b28919a | |||
| b154bb3885 | |||
| 46d6efa962 | |||
| f46ccbae1c | |||
| 772fad18b4 | |||
| 8199774405 | |||
| 66ba06875e | |||
| 25ccb5c75b | |||
| 22e7a1b0e7 | |||
| a9c521a48a | |||
| 8fc757aa61 | |||
| 39050c333f | |||
| 19e5182228 | |||
| 5f321b968a | |||
| 5e28919b39 | |||
| 7bca76e7b6 | |||
| 62a16d2d92 | |||
| 8468c471e8 | |||
| 709d2e4498 | |||
| 0441e319f6 | |||
| 48d51023f2 | |||
| 603e9439d3 | |||
| e5ec63967e | |||
| 24644ab90e | |||
| c988f273b1 | |||
| 7704959f48 | |||
| fd6bed2d11 | |||
| 459e7fa602 | |||
| cc6e85cd9b | |||
| 0b1d9e2b15 | |||
| 10acaec33d | |||
| 1fcb17730e | |||
| d75c4e8e39 | |||
| 263ec2917b | |||
| be171304f5 | |||
| 4aec3061d5 | |||
| e643d38f58 | |||
| 6bf2a84ccd | |||
| 75da6767d3 | |||
| 8ac88ee536 | |||
| 0a8784cc1b | |||
| 7156b25957 | |||
| 022accfa7b | |||
| c457520ae4 | |||
| 9ca931e148 | |||
| c77ff913f0 | |||
| 4a3551254f | |||
| 0b7489f74d | |||
| 3dac5d3b44 | |||
| a99f2473b6 | |||
| ca46a93171 | |||
| 85a2b95428 | |||
| 75dec016eb | |||
| a0a9d5f5e4 | |||
| fc1e1487c7 |
@@ -87,8 +87,8 @@ id = "contract-drift-check-v1"
|
||||
canonical_source = "corviduo-project-template"
|
||||
canonical_path = "scripts/contract_drift_check.py"
|
||||
consumer_path = "scripts/contract_drift_check.py"
|
||||
pinned_sha256_16 = "23271287ac488da4"
|
||||
pinned_at = "2026-05-17T05:30:00+00:00"
|
||||
pinned_sha256_16 = "2659a17a65704b66"
|
||||
pinned_at = "2026-07-12T08:39:35+00:00"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Worldtree Conversation-API surface (vendored from ~/development/Worldtree).
|
||||
@@ -104,8 +104,8 @@ id = "worldtree-conversation-api-openapi-v2"
|
||||
canonical_source = "Worldtree"
|
||||
canonical_path = "docs/conversation-api-openapi.json"
|
||||
consumer_path = "docs/conversation-api-openapi.json"
|
||||
pinned_sha256_16 = "dbdf4e24c8b06c92"
|
||||
pinned_at = "2026-06-30T22:25:56+00:00"
|
||||
pinned_sha256_16 = "36148179601453a0"
|
||||
pinned_at = "2026-07-06T16:09:05+00:00"
|
||||
|
||||
[[pins]]
|
||||
id = "worldtree-conversation-api-sse-events-v1"
|
||||
@@ -120,6 +120,87 @@ id = "worldtree-conversation-api-spec-v1"
|
||||
canonical_source = "Worldtree"
|
||||
canonical_path = "docs/conversation-api-spec.md"
|
||||
consumer_path = "docs/conversation-api-spec.md"
|
||||
pinned_sha256_16 = "2d111a3b8322b7d1"
|
||||
pinned_at = "2026-06-30T22:25:56+00:00"
|
||||
pinned_sha256_16 = "2d73d50b8680b893"
|
||||
pinned_at = "2026-07-13T07:54:05+00:00"
|
||||
tolerate_drift = true # prose reference; OpenAPI+SSE are the gates
|
||||
|
||||
# Worldtree persona render canons (d2) — the deterministic affect->NL the agent is
|
||||
# context-injected. The web persona pane renders mood + relationship-directive BYTE-EXACT
|
||||
# from these (via the flat src/ratatoskr/web/static/persona_render_canon.json, regenerated
|
||||
# by scripts/build_persona_canon.py). Drift here => rerun that regen with Worldtree's venv.
|
||||
[[pins]]
|
||||
id = "worldtree-persona-mood-render-canon-v1"
|
||||
canonical_source = "Worldtree"
|
||||
canonical_path = "core/persona/canon/d2-mood-render-canon-v1.json"
|
||||
consumer_path = "docs/vendor/worldtree-persona-canon/d2-mood-render-canon-v1.json"
|
||||
pinned_sha256_16 = "e2f124fed3ee8d42"
|
||||
pinned_at = "2026-07-01T21:00:00+00:00"
|
||||
|
||||
[[pins]]
|
||||
id = "worldtree-persona-d2-render-canon-v1"
|
||||
canonical_source = "Worldtree"
|
||||
canonical_path = "core/persona/canon/d2-render-canon-v1.json"
|
||||
consumer_path = "docs/vendor/worldtree-persona-canon/d2-render-canon-v1.json"
|
||||
pinned_sha256_16 = "606bba5fdcc60b6b"
|
||||
pinned_at = "2026-07-01T21:00:00+00:00"
|
||||
|
||||
# Worldtree affect-egress consumer reference — the authoritative DELIVERED-on-wire vs
|
||||
# HIDDEN (system-prompt-only) classification for the Tier-3 affect surface ratatoskr
|
||||
# consumes, + the reconstruction rules. The web console's "context injection" panel
|
||||
# reconstructs the hidden strings from this + the d2 canons. tolerate_drift: prose
|
||||
# reference (the render-canon JSONs are the strict gates). worldtree-dev co-signs +
|
||||
# pings ratatoskr-dev on any change (esp. the pending we-framing conditional).
|
||||
[[pins]]
|
||||
id = "worldtree-affect-egress-consumer-reference-v1"
|
||||
canonical_source = "Worldtree"
|
||||
canonical_path = "docs/affect-egress-consumer-reference.md"
|
||||
consumer_path = "docs/vendor/worldtree-persona-canon/affect-egress-consumer-reference.md"
|
||||
pinned_sha256_16 = "b2406e237df00dcb"
|
||||
pinned_at = "2026-07-13T07:54:05+00:00"
|
||||
tolerate_drift = true # prose reference; the d2 render-canon JSONs are the gates
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Brokkr R34/R35 persona-prompt-framing reference (the character-self-report
|
||||
# reframe ratatoskr consumes: the authored psychological_profile is the prose
|
||||
# lens the Worldtree self-report producer reads for affect + memory salience).
|
||||
# Vendored for reference alongside the Worldtree affect/memory surfaces.
|
||||
# tolerate_drift: prose reference, not a machine gate — brokkr-smithy-dev owns
|
||||
# it and pings ratatoskr-dev on canonical changes. The authoring-spec GOVERNS on
|
||||
# any conflict with the parameter distillation.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
[[pins]]
|
||||
id = "brokkr-psych-profile-authoring-spec-v1"
|
||||
canonical_source = "brokkr-smithy"
|
||||
canonical_path = "research/R34-persona-prompt-framing/deliverables/psych-profile-authoring-spec.md"
|
||||
consumer_path = "docs/vendor/brokkr-r34-psych-profile/psych-profile-authoring-spec.md"
|
||||
pinned_sha256_16 = "4545a108d9fb6cc3"
|
||||
pinned_at = "2026-07-13T00:00:00+00:00"
|
||||
tolerate_drift = true # prose reference; brokkr-smithy-dev owns + pings on change
|
||||
|
||||
[[pins]]
|
||||
id = "brokkr-psych-profile-parameters-v1"
|
||||
canonical_source = "brokkr-smithy"
|
||||
canonical_path = "research/R34-persona-prompt-framing/deliverables/psych-profile-parameters.md"
|
||||
consumer_path = "docs/vendor/brokkr-r34-psych-profile/psych-profile-parameters.md"
|
||||
pinned_sha256_16 = "17157c82771aeeee"
|
||||
pinned_at = "2026-07-13T00:00:00+00:00"
|
||||
tolerate_drift = true # parameter distillation; authoring-spec governs on conflict
|
||||
|
||||
[[pins]]
|
||||
id = "soong-lab-export-contract-v1"
|
||||
canonical_source = "soong-lab"
|
||||
canonical_path = "docs/contracts/export.contract.md"
|
||||
consumer_path = "docs/vendor/soong-lab-bundle/export.contract.md"
|
||||
pinned_sha256_16 = "bbd8fcf0cc7bc535"
|
||||
pinned_at = "2026-07-14T17:15:44+00:00"
|
||||
tolerate_drift = true # soong-lab-dev owns the bundle format + pings ratatoskr-dev on change
|
||||
|
||||
[[pins]]
|
||||
id = "soong-lab-importer-contract-v1"
|
||||
canonical_source = "soong-lab"
|
||||
canonical_path = "docs/contracts/importer.contract.md"
|
||||
consumer_path = "docs/vendor/soong-lab-bundle/importer.contract.md"
|
||||
pinned_sha256_16 = "777b1764c8eb2cb7"
|
||||
pinned_at = "2026-07-14T17:15:44+00:00"
|
||||
tolerate_drift = true # soong-lab-dev owns the bundle format + pings ratatoskr-dev on change
|
||||
|
||||
@@ -121,3 +121,7 @@ graphify-out/*
|
||||
*.db
|
||||
*.db-shm
|
||||
*.db-wal
|
||||
|
||||
# Node deps (Playwright for web-UI DOM verification — see persistent-memory)
|
||||
node_modules/
|
||||
package-lock.json
|
||||
|
||||
+9
-7
@@ -7,16 +7,18 @@ documents the pin, the vendored artifacts, and the bump procedure.
|
||||
|
||||
| Field | Value |
|
||||
|---|---|
|
||||
| Worldtree git SHA | `5810a26b38a5ea6630892f9a39756f57c5b7b41e` |
|
||||
| Worldtree HEAD message | `memory: snapshot — v1.0.0b2 shipped complete (demo + personal green); consumer loop closed` |
|
||||
| Pinned on | 2026-06-30 |
|
||||
| Pinned by | ratatoskr-dev (v1 coverage-audit — re-pin to the FROZEN OpenAPI 2.2.0 + SSE schema) |
|
||||
| Worldtree version at pin | `v1.0.0b2` |
|
||||
| Worldtree git SHA | `c9e59ec` |
|
||||
| Worldtree HEAD message | `docs: document Tier-3 persona/memory schemas + persona_state SET body (OpenAPI 2.3.0)` |
|
||||
| Pinned on | 2026-07-06 |
|
||||
| Pinned by | ratatoskr-dev (re-vendor prose markdown — Tier-3 persona/memory/persona_state consumer shapes) |
|
||||
| Worldtree version at pin | `v1.0.0b22` |
|
||||
|
||||
## Pin history
|
||||
|
||||
| Date | SHA | Version | Notable deltas consumed |
|
||||
|---|---|---|---|
|
||||
| 2026-07-06 | `c9e59ec` | v1.0.0b22 | **Re-vendor the prose markdown — Tier-3 consumer shapes documented.** `c9e59ec` (docs-only, OpenAPI byte-unchanged vs `879cefe`) adds `docs/conversation-api-spec.md` § "Tier 3 — Consumer-defined agents": the persona / memory / persona_state SET-body shapes that serialize as freeform `Any` in the OpenAPI (so prose is their source of truth). Drove a consumer fix: `--set-persona-pad` now sends the canonical `{pad:{pleasure,arousal,dominance}}` named dict (was `{pad:[list]}`) — #317, `v0.19.7`. Foot-guns encoded: persona.ocean single-letter `{O,C,E,A,N}` on `/agents/define` (spelled-out → 422, the #348 mismatch) vs spelled-out on `POST /characters`; memory `{embedder_version, tier3_dreaming}`, stm_* deprecated, allows_world_scope removed→422; only `valence` still 422s. `pin:`-only for the markdown; the `v0.19.7` bump rode the persona_state code fix. |
|
||||
| 2026-07-06 | `879cefe` | v1.0.0b22 | **Re-vendor OpenAPI 2.2.0→2.3.0 — Worldtree shipped #347 authored-history-write.** One new REST path-group: `POST /sessions/{session_id}/history` (the authored-history-write primitive) + the `AuthoredTurnResponse` schema (openapi path count 40→41). #347 is **OpenAPI-only** — the prose `conversation-api-spec.md` + server `conversation_api.contract.md` are byte-unchanged since the 5810a26 pin (empty `git log` delta), so those `tolerate_drift` pins stay clean; the SSE schema is unchanged (#347 is event-silent by design). **Consumer side NOT yet built** — `POST /sessions/{id}/history` is a fresh in-scope ⬜ gap in `docs/coverage-map.md` (re-opens the v1 coverage-audit with exactly one gap; Heimdall-gated hide-existence → consumer treats 404 as feature-absent). `pin:`-only, no version bump. |
|
||||
| 2026-06-30 | `5810a26` | v1.0.0b2 | **Re-pin to Worldtree's FROZEN v1 surface (#326), as part of the v1 coverage-audit.** Vendored the machine-readable artifacts — `conversation-api-openapi.json` (OpenAPI **2.2.0**, 40 path-groups) + `conversation-api-sse-events.schema.json` (11 events) — now the **authoritative drift gates** (pinned in `.corviduo-canonicals.toml`, CI-checked by `canonical_drift.py`). The prose `conversation-api-spec.md` is **byte-identical** to the v0.35.16 pin (last WT markdown edit 2026-05-31), kept as the human reference (`tolerate_drift`). b2 deltas already consumed in code: 409/503 eager turn-launch statuses (#331, v0.18.3/.4) + the unified error envelope (#328). 7 endpoints documented only in the OpenAPI, not the prose, all classified in `docs/coverage-map.md`: `admin/keys/bulk`, `admin/persona/{archive,erase}`, `admin/usage`, `embed`, `judgments`, `me/usage`. No client-breaking change — `pin:`-only, no version bump. |
|
||||
| 2026-06-17 | `f1b59f8` | v0.35.16 | **#297 + #298/#299 — Worldtree adopts the bifrost v0.6 scope wire (emits `scope_any`/`scope_all`) + client-side per-scope-value union recall. With our v0.17.6 provider this closes cold cross-session recall end-to-end.** Catch-up bump (v0.29.0→v0.35.16). Intervening client-facing deltas reviewed, none break our consumer: #211 agent rename (`saga`→`echo`, `actor`→`mask` — slugs only); #245 `end_user_id` persistence + memory-scope resolver; #187/#188/#219 Tier-3 define/PATCH policy (additive); `bifrost` binding field + `ephemeral_does_not_accept_bifrost` 422 now documented (the #17 surface). Error codes stable; no ratatoskr code change required. |
|
||||
| 2026-05-25 | `da93ca7` | v0.28.0 | #204 — new SSE event `affect_update` (current/scheduled), new endpoint `GET /agents/{id}/persona_state`, auth-model doc edits |
|
||||
@@ -26,12 +28,12 @@ documents the pin, the vendored artifacts, and the bump procedure.
|
||||
|
||||
**Authoritative (FROZEN, machine-readable — the drift gates):**
|
||||
|
||||
- `docs/conversation-api-openapi.json` — copy of `Worldtree/docs/conversation-api-openapi.json` (OpenAPI `info.version` **2.2.0**). The frozen v1 REST wire (40 path-groups). Pinned `worldtree-conversation-api-openapi-v2` in `.corviduo-canonicals.toml`; drift gated by `canonical_drift.py`.
|
||||
- `docs/conversation-api-openapi.json` — copy of `Worldtree/docs/conversation-api-openapi.json` (OpenAPI `info.version` **2.3.0**). The frozen v1 REST wire (41 path-groups; 2.3.0 added `POST /sessions/{session_id}/history` per #347). Pinned `worldtree-conversation-api-openapi-v2` in `.corviduo-canonicals.toml`; drift gated by `canonical_drift.py`.
|
||||
- `docs/conversation-api-sse-events.schema.json` — copy of `Worldtree/docs/conversation-api-sse-events.schema.json`. The frozen SSE event schema (11 discriminated event types). Pinned `worldtree-conversation-api-sse-events-v1`.
|
||||
|
||||
**Reference (prose; allowed to lag — `tolerate_drift`):**
|
||||
|
||||
- `docs/conversation-api-spec.md` — copy of `Worldtree/docs/conversation-api-spec.md` at the pinned SHA. The **client-facing prose narrative**. Byte-frozen at v0.35.16-era content (last WT edit 2026-05-31); the OpenAPI/SSE JSON above are the source of truth where they diverge. Pinned `worldtree-conversation-api-spec-v1` (tolerate_drift).
|
||||
- `docs/conversation-api-spec.md` — copy of `Worldtree/docs/conversation-api-spec.md` at the pinned SHA. The **client-facing prose narrative**. Re-vendored at `c9e59ec` (2026-07-06) to carry the § "Tier 3 — Consumer-defined agents" subsections (persona/memory/persona_state SET body) that serialize as freeform `Any` in the OpenAPI JSON — so the **prose is the source of truth for those consumer shapes** (e.g. persona.ocean single-letter `{O,C,E,A,N}` on `/agents/define`; `POST /sessions/{id}/persona_state` body `{pad:{pleasure,arousal,dominance}}`). Elsewhere the OpenAPI/SSE JSON above remain authoritative. Pinned `worldtree-conversation-api-spec-v1` (tolerate_drift).
|
||||
- `docs/conversation_api.contract.md` — copy of `Worldtree/docs/contracts/conversation_api.contract.md` at the pinned SHA (byte-identical at b2 — server contract unchanged since the v0.35.16 pin). The **server-side contract** including INV-001..INV-052 and amendments. Useful for understanding load-bearing server invariants (e.g., INV-014 turn-id-public, INV-046 admin-events-envelope-stable, INV-049 admin-events-pii-discipline) when designing client behavior against them. Not in the canonical manifest (reference-only).
|
||||
|
||||
Both files are vendored — they reflect Worldtree at the pinned SHA, not
|
||||
|
||||
@@ -116,9 +116,17 @@ interpreted.
|
||||
to the reference `_matches_scope`. (`scope_any` is the union-visibility primitive that
|
||||
resolves the #295/#297 silent-zero — a subset-scoped chunk now recalls via an OR member.)
|
||||
- **INV-006** [hard]: **Capabilities match implementation** (advertise-⇒-implement).
|
||||
`describe_store` advertises ONLY what v1 implements: `relational_edges_supported=False`,
|
||||
`describe_store` advertises ONLY what is implemented: `relational_edges_supported=False`,
|
||||
`atomic_supersede_supported=False`, `transaction_supported=False`,
|
||||
`optimistic_locking_supported=True`, `filterable_metadata_fields=[]`.
|
||||
`optimistic_locking_supported=True`, `filterable_metadata_fields=[]`,
|
||||
**`sortable_chunk_fields=[{"name": "updated_at", "type": "timestamp"}]`** (the ONLY
|
||||
globally-sortable field; gates `scan`'s sort at the bifrost dispatch `_validate_scan_sort`
|
||||
AND Worldtree's #349 person-prime Branch-A `"updated_at" in caps.sort_fields_supported` —
|
||||
advertising it is what lights up turn-1 durable-fact injection). Both `name` AND `type`
|
||||
are REQUIRED by the bifrost `handshake_response` `SortableChunkField` schema
|
||||
(`additionalProperties:false`) — omitting `type` fails wire-schema validation and breaks
|
||||
the ENTIRE handshake (memory + affect bind), not just the sort; `type` is advisory-only
|
||||
(the wire never interprets it).
|
||||
(`transaction_supported` is the bifrost **wire-level** multi-op transaction
|
||||
capability — NOT our internal SQLite transactions, which we use for atomic
|
||||
batches.) The client gates the gated verbs off these.
|
||||
@@ -127,6 +135,41 @@ interpreted.
|
||||
`InvalidArguments` (mirrors the reference).
|
||||
- **INV-008** [hard]: The store is REQUIRED (`build_memory_app(store=None)` raises);
|
||||
identity/scope/actor come from `ctx`, never call args.
|
||||
- **INV-009** [hard]: **`scan` is LIVE-only.** `scan` returns ONLY live chunks —
|
||||
superseded / tombstoned / any non-live governance state is EXCLUDED server-side. This
|
||||
is load-bearing because Worldtree's person-prime requests `lifecycle_state="live"` but
|
||||
that filter does NOT ride the scan wire today and the client does not re-check it
|
||||
(worldtree-dev flagged the adapter gap); server-side live-only is authoritative, so a
|
||||
dead fact can never inject. The additive `lifecycle_state` scan arg, when present, is
|
||||
honored but never relied upon.
|
||||
- **INV-010** [hard]: **`scan` is globally ordered before pagination.** The FULL
|
||||
scope-filtered live set is ordered by `(sort.field, direction)` GLOBALLY before the
|
||||
`limit` page is taken — never page-local. Missing sort value sorts LAST; ties broken by
|
||||
`chunk_id` (stable). A single `limit`-page returns the N globally-newest (for
|
||||
`updated_at desc`), matching bifrost's cross-pagination conformance negative. The sort
|
||||
field is indexed (`json_extract(record_json, '$.updated_at')`) so the read stays within
|
||||
person-prime's 500 ms fail-open budget.
|
||||
- **Cursor is v1-provisional (KNOWN DEVIATION — offset, not snapshot).** The cursor is a
|
||||
bare integer offset into the re-derived global order. This is CORRECT and conformant for
|
||||
the **single-page** person-prime call (`cursor=None`), which is the only shipped consumer.
|
||||
It **diverges from bifrost's protocol snapshot-cursor contract on multi-page continuation**:
|
||||
the dispatch engine (`bifrost.memory` scan branch) drops the `sort` arg on a cursor
|
||||
continuation because "the cursor's snapshotted order is authoritative", and maps
|
||||
`ScanCursorExpired → 410`. Our offset cursor (a) does NOT snapshot the order — a page taken
|
||||
after a concurrent write can duplicate/drop rows relative to the first page (heid-bug-hunt
|
||||
2026-07-15, all 3 arms), and (b) never raises `ScanCursorExpired`. The `global_before_paginate`
|
||||
/ cursor test asserts **static-store** behavior only. The durable/conformant fix is to adopt
|
||||
the reference `InMemoryMemoryStore`'s snapshot-cursor semantics (opaque token + frozen ordered
|
||||
id-list + TTL + `ScanCursorExpired`); DEFERRED pending bifrost-dev's ruling on the conformance
|
||||
gap (scan/cursor has NO conformance coverage today, so a non-snapshot cursor passes). Routed
|
||||
to bifrost-dev 2026-07-15.
|
||||
- **INV-011** [hard]: **`mark_superseded` retires via a top-level `superseded` flag; `_is_live`
|
||||
recognizes it.** `mark_superseded` sets top-level `superseded=True` (+ `superseded_by`) on the
|
||||
record, mirroring the reference `_mark_lifecycle` (NOT a `verbatim.governance_state` change). So
|
||||
`_is_live` MUST short-circuit on `record.get("superseded") is True` (in addition to its existing
|
||||
`lifecycle_state` / `verbatim.governance_state` checks) — else a #364-retired chunk would still
|
||||
scan live. Retirement is NON-destructive: `get`/`get_many` still return superseded chunks
|
||||
(recoverable). `search` is NOT filtered (matches the reference; WT re-checks liveness client-side).
|
||||
|
||||
## Concurrency
|
||||
|
||||
@@ -163,7 +206,7 @@ negotiation, routes). **This contract** owns the store (the basic verbs + SQLite
|
||||
|
||||
## Out of scope (deferred — do NOT flag as drift)
|
||||
|
||||
- **Gated/maintenance verbs:** `upsert_edges`/`get_edges_for`, `scan`, `mark_invalid`/`mark_superseded`, `patch_many`, `atomic_supersede`, lease/checkpoint. Absent + advertised-unsupported.
|
||||
- **Gated/maintenance verbs:** `upsert_edges`/`get_edges_for`, `mark_invalid`, `patch_many`, `atomic_supersede`, lease/checkpoint. Absent (no describe_store cap; hasattr-gated at dispatch as of bifrost 1.1.4 → `unsupported_capability` 400). (`scan` and `mark_superseded` are NO LONGER deferred — `scan` implements #349 person-prime; `mark_superseded` implements Worldtree #364's contradiction retirement, the SOLE supersession verb #364 uses. See their FN specs + INV-009/INV-011.)
|
||||
- **metadata_filter beyond scope:** advertise `filterable_metadata_fields=[]`; a non-empty `metadata_filter` is unsupported in v1 (rejected — see search PRE).
|
||||
- **The combined two-plane server** (guide §7) — separate memory + affect apps in v1.
|
||||
- **Deployment** — dev-box background shell (`ratatoskr-memory-provider`), no systemd/infra.
|
||||
@@ -270,6 +313,47 @@ TESTS:
|
||||
delete_absent [boundary]: unknown id → {"deleted":0}
|
||||
```
|
||||
|
||||
```contract
|
||||
FN mark_superseded(self, ids: list[str], *, superseded_by: str | None = None, reason: str | None = None) -> dict
|
||||
BRIEF: Worldtree #364 retirement — mark chunks superseded so scan (live-only) excludes them. Mirrors the reference _mark_lifecycle: sets TOP-LEVEL fields on the record; NON-destructive (get still returns them, recoverable). The SOLE supersession verb #364 uses (dispatch: bifrost/memory.py mark_superseded branch; args {ids:[...], superseded_by, reason}).
|
||||
PRE: [PRE-001 hard] ids is a list of chunk ids (WT sends singletons, one call per retired chunk)
|
||||
POST: [POST-001 return_value] {"marked": N} where N = ids that existed (unknown ids skipped, never error) -- assert
|
||||
POST: [POST-002 state_change] each existing chunk gets top-level `superseded=True` + `superseded_by` (when not None) + `superseded_reason` (when not None); revision incremented; mirrors reference _mark_lifecycle (only non-None fields written) -- assert
|
||||
POST: [POST-003 return_value] a superseded chunk is EXCLUDED from `scan` (INV-009 via _is_live's top-level `superseded` check, INV-011) but STILL returned by `get`/`get_many` (non-destructive) -- assert
|
||||
STEPS:
|
||||
1. [sequential, flexibility=indicative] FOR each id present: load record_json, set superseded=True (+ superseded_by / superseded_reason when not None), UPDATE record_json + revision+1; count
|
||||
2. [cleanup] RETURN {"marked": count}
|
||||
TESTS:
|
||||
mark_retires_from_scan [happy,tracer]: upsert 3 live; mark_superseded([id2], superseded_by="x"); scan → the 2 non-superseded only (id2 excluded); id2 record has superseded=True + superseded_by="x"
|
||||
mark_get_still_returns [scenario]: a superseded chunk is STILL returned by get (non-destructive/recoverable)
|
||||
mark_unknown_id_noop [boundary]: mark_superseded(["nope"]) → {"marked":0}
|
||||
mark_no_superseded_by [boundary]: mark_superseded([id], superseded_by=None) → superseded=True set, no superseded_by key written (only non-None fields)
|
||||
mark_parity_vs_reference [scenario]: identical mark_superseded envelope vs InMemoryMemoryStore → same top-level superseded/superseded_by field shape (#195)
|
||||
```
|
||||
|
||||
```contract
|
||||
FN scan(self, *, scope_all: dict | None = None, scope_any: list | None = None, cursor: str | None = None, limit: int, sort: dict | None = None, lifecycle_state=None) -> dict
|
||||
BRIEF: Query-LESS paginated LIVE-chunk scan, globally ordered by an advertised sort field (updated_at) — the #349 person-prime turn-1 durable-fact injection primitive (no query vector, unlike search). Returns {records, cursor}.
|
||||
PRE: [PRE-001 hard] limit is a positive int -- else InvalidArguments
|
||||
PRE: [PRE-002 hard] scope_all/scope_any shape + lattice-validated via _validate_scope (identical to search PRE-003) -- else InvalidArguments / InvalidFilter
|
||||
PRE: [PRE-003 hard] sort, when present, is {field, direction}: field ∈ the advertised sortable_chunk_fields names ("updated_at"), direction ∈ {asc,desc}. The bifrost dispatch layer (_validate_scan_sort) is the enforcement gate; an unadvertised/malformed sort → InvalidArguments — NEVER a silent unsorted fallback
|
||||
POST: [POST-001 return_value] {records: [<verbatim chunk wire records, same shape as a search hit's chunk>], cursor: <opaque next-page str | None>}; ≤ limit records; each record carries updated_at + agent_id + subject{type,id} + worldtree_scope (the fields person-prime's client _scan_filter_matches keys on — a record missing any is silently dropped client-side) -- assert
|
||||
POST: [POST-002 return_value] LIVE-only — returns ONLY live chunks; superseded/tombstoned excluded server-side (INV-009)
|
||||
POST: [POST-003 return_value] GLOBAL-order — the FULL scope-filtered live set is ordered by (sort.field, direction) GLOBALLY before the limit page; missing value LAST; chunk_id tiebreak (INV-010)
|
||||
STEPS:
|
||||
1. [setup] validate limit (>0) + scope (as search); sort ← the dispatch-validated {field,direction}
|
||||
2. [sequential, flexibility=indicative] SELECT scope-filtered LIVE chunks ordered by the indexed sort field (json_extract(record_json,'$.updated_at')) in `direction`, missing-last, chunk_id tiebreak, GLOBALLY; apply cursor offset; take limit
|
||||
3. [cleanup] RETURN {records: verbatim chunks, cursor: next-page-or-None}
|
||||
TESTS:
|
||||
scan_recency [happy,tracer]: upsert 4 live chunks w/ distinct updated_at; scan(scope_all={end_user}, limit=3, sort={field:updated_at,direction:desc}) → the 3 newest, newest-first
|
||||
global_before_paginate [scenario]: 5 chunks, limit=2 → page-1 = the 2 globally-newest; the cursor page continues the GLOBAL order, not a page-local re-sort (INV-010; bifrost cross-pagination conformance)
|
||||
live_only [adversarial]: a superseded/tombstoned chunk is NEVER returned even if it is the newest (INV-009)
|
||||
scope_isolation [adversarial]: scope_all one end_user → never returns another partition's chunk (INV-005 applies to scan)
|
||||
unadvertised_sort [adversarial]: sort.field ∉ sortable_chunk_fields → InvalidArguments at dispatch (never silent unsorted)
|
||||
person_prime_record_shape [scenario]: each record carries agent_id + subject{type,id} + worldtree_scope + updated_at + verbatim/distillate — the _scan_filter_matches keys (else the client silently drops it)
|
||||
parity_vs_reference [scenario]: identical scan envelopes vs InMemoryMemoryStore → same ordered chunk_ids/shape (#195)
|
||||
```
|
||||
|
||||
```contract
|
||||
FN build_memory_provider_app(store: RatatoskrMemoryStore, heimdall_key: bytes, consumer_id: str = "ratatoskr") -> Starlette
|
||||
BRIEF: Wire JwtVerifier + registration; hand the store to bifrost's build_memory_app.
|
||||
|
||||
@@ -0,0 +1,142 @@
|
||||
---
|
||||
contract_version: "2.1"
|
||||
module: "ratatoskr.first_message"
|
||||
purpose: "Per-agent authored first-message presets — seed an agent's opening as a #347 authored turn-0 onto new sessions (CLI + web), the durable replacement for a system-prompt startup instruction."
|
||||
touches:
|
||||
- src/ratatoskr/first_message.py
|
||||
- tests/test_first_message.py
|
||||
scope: >
|
||||
Per-agent authored first-message presets (Worldtree #347 consumer feature).
|
||||
When a new session is created for an agent that has a preset opening, seed it
|
||||
as a #347 authored first-message (POST /sessions/{id}/history, author=assistant,
|
||||
seq-0) so the session opens in-character before the user speaks — the durable
|
||||
replacement for a system-prompt "startup" instruction. Two entry points:
|
||||
`preset_for` (lookup) and `seed_preset_first_message` (best-effort seed).
|
||||
Consumed by ratatoskr.cli (the `--new` session path) and ratatoskr.web.server
|
||||
(the POST /api/sessions endpoint). Depends on ratatoskr.sessions
|
||||
(write_authored_history + its exceptions); no core.* / worldtree.* imports.
|
||||
depends_on:
|
||||
- "httpx"
|
||||
- "ratatoskr.sessions"
|
||||
used_by:
|
||||
- "ratatoskr.cli"
|
||||
- "ratatoskr.web.server"
|
||||
language: "python"
|
||||
complexity: "low"
|
||||
estimated_loc: 60
|
||||
confidence: 0.9
|
||||
assumptions:
|
||||
- "write_authored_history (contract #2 amendment 2026-07-06) is the seed primitive: 200/201 → ack dict, 404 → AuthoredHistoryUnavailable (hide-existence), other non-2xx → SessionApiFailed."
|
||||
- "The preset registry is a static in-module dict keyed by agent_id; editing it is how an operator tunes an agent's opening. Seeded with ratatoskr:sindra only."
|
||||
- "Auto-seed is BEST-EFFORT and MUST NOT block session creation: an instance without the session.history.write grant returns the hide-404, which is swallowed (session opens with no seeded greeting)."
|
||||
---
|
||||
|
||||
# First-message presets — authored openings on session-create (#347)
|
||||
|
||||
## Context
|
||||
|
||||
`ratatoskr.first_message` holds per-agent authored-opening presets and seeds them
|
||||
onto new sessions via the #347 authored-history-write primitive. It is the
|
||||
durable form of "give an agent a first message": instead of a system-prompt
|
||||
`Startup:` instruction (a workaround for the pre-#347 world where the assistant
|
||||
could not author turn-0), the opening lives as a real seeded assistant turn-0.
|
||||
|
||||
Consumed at both session-create sites — `ratatoskr.cli._amain` (the `--new` path)
|
||||
and `ratatoskr.web.server._create_session_endpoint` (POST /api/sessions) — so
|
||||
every new session for a preset agent opens in-character regardless of surface.
|
||||
|
||||
## Data flow
|
||||
|
||||
**In:** a live `httpx.AsyncClient` (caller-owned, base_url + bearer set), a fresh
|
||||
`session_id`, and the bound `agent_id`.
|
||||
|
||||
**Out:** on a preset agent, one `POST /sessions/{session_id}/history` (author=assistant,
|
||||
the preset text, per-content idempotency key). Returns the seeded content on
|
||||
success, else `None`.
|
||||
|
||||
**Side effects:** at most one outbound authored-history write; never raises to the
|
||||
caller (best-effort).
|
||||
|
||||
## Invariants
|
||||
|
||||
- **INV-001 [hard]**: `seed_preset_first_message` NEVER raises (the sole exception is
|
||||
`asyncio.CancelledError`, which propagates — cancellation is not a seed failure) and
|
||||
NEVER blocks session creation. It soft-guards its inputs (a bad arg returns `None`,
|
||||
not `AssertionError`), bounds the write with `asyncio.wait_for(_SEED_TIMEOUT_S)` so a
|
||||
stalled `/history` can't hang the create path, and swallows EVERY other exception (the
|
||||
hide-404, `SessionApiFailed`, `httpx.HTTPError`, `TimeoutError`, and any unexpected
|
||||
error) → `None`. The `broad-except` is deliberate: this helper is wired INTO three
|
||||
session-create paths, so any escape would abort a create that already succeeded.
|
||||
- **INV-002 [hard]**: a no-preset agent issues ZERO HTTP (early return before any
|
||||
request).
|
||||
- **INV-003 [hard]**: the seed body is the preset text verbatim, author="assistant",
|
||||
with a per-content idempotency key (`"ratatoskr-preset-" + sha256(text)[:12]`), so
|
||||
a repeat seed of the same session+preset is an idempotent 200 replay, never a
|
||||
duplicate turn.
|
||||
- **INV-004 [hard]**: no `core.*` / `worldtree.*` imports (reference-consumer
|
||||
boundary; verified by `tests/test_no_worldtree_imports.py`, which rglobs every
|
||||
`.py` under `src/ratatoskr/` — this module included, so no per-module import
|
||||
test is needed here).
|
||||
|
||||
## Out of scope
|
||||
|
||||
- **Multi-turn / scripted openers.** v1 seeds exactly one assistant turn-0. A
|
||||
multi-message opening scene is a future concern.
|
||||
- **Runtime/remote preset config.** The registry is an in-module dict; no file/DB/env
|
||||
loading. Add that only when a second consumer needs operator-editable presets.
|
||||
- **Non-assistant authors.** v1 is author=assistant only (matches #347 v1); a
|
||||
user/system opener is deferred with the #347 engine surface.
|
||||
- **TUI-only surfaces.** Both real session-create paths (CLI + web) are wired; the
|
||||
bare-TUI picker resumes existing sessions (no create), so it needs no seed.
|
||||
|
||||
---
|
||||
|
||||
```contract
|
||||
FN preset_for(agent_id: str) -> str | None
|
||||
BRIEF: Return the authored first-message preset for agent_id, or None when the agent has no preset. Pure dict lookup over FIRST_MESSAGE_PRESETS.
|
||||
PRE: [PRE-001 hard] agent_id is a non-empty str -- assert agent_id and isinstance(agent_id, str)
|
||||
POST: [POST-001 return_value] returns FIRST_MESSAGE_PRESETS.get(agent_id) (str for a preset agent, None otherwise)
|
||||
STEPS:
|
||||
1. [setup, prescriptive] assert PRE-001
|
||||
2. [sequential, prescriptive] RETURN FIRST_MESSAGE_PRESETS.get(agent_id)
|
||||
TESTS:
|
||||
preset_hit [happy]: preset_for("ratatoskr:sindra") is a non-empty str
|
||||
preset_miss [happy]: preset_for("mimir") is None
|
||||
empty_agent_id [adversarial]: preset_for("") → AssertionError
|
||||
|
||||
FN seed_preset_first_message(client: httpx.AsyncClient, session_id: str, agent_id: str) -> str | None
|
||||
BRIEF: Best-effort seed of an agent's preset opening as a #347 authored first-message on session_id. If agent_id has a preset, POST it via write_authored_history (author=assistant, per-content idempotency key, the await bounded by asyncio.wait_for(_SEED_TIMEOUT_S)) and return the seeded content; on no-preset, a malformed input, OR ANY exception except asyncio.CancelledError, return None WITHOUT raising. Never raises (except CancelledError, which propagates) and never blocks session creation — it is wired into three create paths.
|
||||
PRE: [PRE-001 hard] client is not None -- soft-guarded: return None (NOT assert) if violated, so a wiring bug can't crash the create path (INV-001)
|
||||
PRE: [PRE-002 hard] session_id is a non-empty str -- soft-guarded: return None if violated
|
||||
PRE: [PRE-003 hard] agent_id is a non-empty str -- soft-guarded: return None if violated (also guards FIRST_MESSAGE_PRESETS.get against a non-hashable/non-str id)
|
||||
POST: [POST-001 return_value] preset agent + successful write → returns the preset text; no-preset, malformed input, OR any swallowed failure → None
|
||||
POST: [POST-002 side_effect] a no-preset / malformed-input call issues ZERO HTTP; a preset agent issues exactly one POST /sessions/{session_id}/history with body author="assistant", content=preset, idempotency_key="ratatoskr-preset-"+sha256(preset)[:12], the await bounded by _SEED_TIMEOUT_S so a stalled response cannot block
|
||||
ERROR_ROUTING:
|
||||
asyncio.CancelledError:
|
||||
local_handling: RE-RAISE (cancellation is not a seed failure; never swallow it — and it is a BaseException, so `except Exception` would miss it anyway)
|
||||
flow_control: propagate
|
||||
state_recovery: n/a
|
||||
any other Exception (hide-404 AuthoredHistoryUnavailable, SessionApiFailed 409/422/etc., httpx.HTTPError, TimeoutError from wait_for, any unexpected error):
|
||||
local_handling: swallow; return None
|
||||
flow_control: continue (never blocks session create)
|
||||
state_recovery: session opens with no seeded greeting
|
||||
STEPS:
|
||||
1. [setup, prescriptive] Soft-guard: IF agent_id is not a non-empty str: RETURN None (before any dict lookup — guards a non-hashable id)
|
||||
2. [sequential, prescriptive] content = FIRST_MESSAGE_PRESETS.get(agent_id); IF content is None: RETURN None (INV-002 — zero HTTP)
|
||||
3. [sequential, prescriptive] Soft-guard: IF client is None OR session_id is not a non-empty str: RETURN None
|
||||
4. [sequential, prescriptive] key = "ratatoskr-preset-" + sha256(content utf-8)[:12]
|
||||
5. [sequential, prescriptive] TRY: await asyncio.wait_for(write_authored_history(client, session_id, content=content, idempotency_key=key), timeout=_SEED_TIMEOUT_S)
|
||||
tool: { destructive: false, idempotent: true, read_only: false, open_world: false }
|
||||
6. [branch, prescriptive] EXCEPT asyncio.CancelledError: RAISE; EXCEPT Exception: RETURN None
|
||||
7. [cleanup, prescriptive] RETURN content
|
||||
TESTS:
|
||||
seeds_preset [happy,tracer]: preset agent, mock 201 → returns the preset text; exactly one POST /sessions/{id}/history; body author="assistant" + content=preset + idempotency_key="ratatoskr-preset-"+sha256(preset)[:12]
|
||||
no_preset_zero_http [happy]: agent "mimir" → returns None; NO HTTP issued
|
||||
feature_absent_swallowed [error]: preset agent, mock 404 session_not_found → returns None, no raise
|
||||
session_api_failed_swallowed [error]: preset agent, mock 409 → returns None, no raise
|
||||
transport_error_swallowed [error]: preset agent, mock httpx.ConnectError → returns None, no raise
|
||||
unexpected_exception_swallowed [error]: preset agent, write raises ValueError → returns None, no raise (INV-001 broad never-raise)
|
||||
cancellation_propagates [error]: preset agent, write raises asyncio.CancelledError → RE-RAISED (never swallowed)
|
||||
malformed_agent_id_no_http [adversarial]: agent_id=123 (non-str) OR "" → None; NO HTTP; no raise
|
||||
empty_session_id [adversarial]: session_id="" (preset agent) → None (soft guard); NO HTTP; no raise
|
||||
```
|
||||
@@ -322,7 +322,9 @@ via a `--characters` one-shot lifecycle probe; persona-state write surfaced via
|
||||
wrappers: parsed dict verbatim (or None on 204), any off-status → SessionApiFailed.
|
||||
**Note:** `set_persona_state`'s request body is FREEFORM — the frozen OpenAPI 2.2.0
|
||||
declares no request schema and the prose spec documents only the GET counterpart,
|
||||
so the caller supplies the snapshot shape (`--set-persona-pad` sends `{pad:[…]}`).
|
||||
so the caller supplies the snapshot shape. **Canonical (worldtree-dev prose #317,
|
||||
`c9e59ec`): `{pad:{pleasure,arousal,dominance}}` — a named-key dict, NOT a list;
|
||||
`--set-persona-pad` builds + sends the named dict (each float in [-1,1]).**
|
||||
|
||||
```contract
|
||||
FN list_character_models(client) -> dict[str, Any]
|
||||
@@ -369,6 +371,102 @@ POST: [POST-001 return_value] on 204 returns None; [POST-002 side_effect] outbou
|
||||
STEPS:
|
||||
1. [sequential, prescriptive] resp = await client.post(f"/sessions/{session_id}/persona_state", json=snapshot); IF 204 RETURN None; ELSE RAISE SessionApiFailed
|
||||
TESTS:
|
||||
happy [happy]: 204 → None; body == {"pad":[...]} verbatim
|
||||
happy [happy]: 204 → None; body == {"pad":{"pleasure","arousal","dominance"}} verbatim (canonical named-key dict, #317)
|
||||
non_204 [error]: 422 → SessionApiFailed(422)
|
||||
```
|
||||
|
||||
## Amendment 2026-07-06 — authored-history write (#347, v1 coverage-audit re-open)
|
||||
|
||||
Worldtree shipped #347 (authored-history-write) as OpenAPI 2.3.0: a new
|
||||
`POST /sessions/{session_id}/history` primitive that writes ONE model-visible
|
||||
turn into a session's ledger AS the bound agent, WITHOUT a generation and
|
||||
WITHOUT lived-turn side effects (the SillyTavern "first message"). The re-vendor
|
||||
(2.2.0→2.3.0, pin `879cefe`) re-opened the v1 coverage-audit with this one new
|
||||
in-scope REST path-group; this amendment closes it on the consumer side and also
|
||||
un-defers `GET /sessions/{id}/messages` (previously §Out of scope) as the seed's
|
||||
read-back.
|
||||
|
||||
**Hide-existence (server INV-347-1) — the load-bearing consumer contract.** The
|
||||
`session.history.write` grant is checked FIRST — an ungranted caller (or a
|
||||
non-owner, or an unknown session) gets a 404 **byte-identical** to a genuine
|
||||
`session_not_found`, never a 403/409/422 that would reveal the feature exists.
|
||||
The consumer MUST honor this: treat 404 as **feature-absent**, fall back (a
|
||||
production consumer to a model-generated greeting), and NEVER capability-probe to
|
||||
tell feature-absent from ungranted from session-absent. The wrapper encodes it by
|
||||
raising a DISTINCT `AuthoredHistoryUnavailable` on 404 (NOT `SessionApiFailed`),
|
||||
so a caller branches feature-absent without inspecting a status code.
|
||||
|
||||
**Request body — v1-minimal, wire-pinned by the server.** The frozen OpenAPI 2.3.0
|
||||
exports an empty request schema, but the server pins `AuthoredWriteRequest`
|
||||
(`extra="forbid"`): `{author, content, idempotency_key, effects?,
|
||||
claimed_original_at?}`. v1: `author="assistant"` (only value), `content` (UTF-8,
|
||||
server-bounded at `authored_content_max_bytes`=8192), `idempotency_key` (REQUIRED,
|
||||
per-session dedup), `effects` omitted (== "none"; only value). Because
|
||||
`extra="forbid"`, the wrapper omits `effects`/`claimed_original_at` when None
|
||||
(never sends null). Success is 201 (fresh) OR 200 (idempotent replay,
|
||||
byte-identical body); both return the `AuthoredTurnResponse` `{author,
|
||||
content_chars, injected_at, phase, seq, session_id, turn_id}` verbatim (provenance
|
||||
is audit-only, NEVER on this body — INV-347-7).
|
||||
|
||||
**Assistant-first provider constraint (deferred, inert for the probe).** A
|
||||
create-time first-message makes the assistant seq-0 (assistant-first history);
|
||||
Anthropic-family providers 400 the *next generation*, vLLM/openai_compat tolerate
|
||||
it. The `--seed-first-message` probe seeds but does NOT generate, so the
|
||||
constraint is inert for the probe — a real consumer that then generates must bind
|
||||
an assistant-first-tolerant provider.
|
||||
|
||||
```contract
|
||||
FN write_authored_history(client: httpx.AsyncClient, session_id: str, *, content: str, idempotency_key: str, author: str = "assistant", effects: str | None = None, claimed_original_at: str | None = None) -> dict[str, Any]
|
||||
BRIEF: POST /sessions/{session_id}/history — the #347 authored-history-write primitive (write one model-visible turn as the bound agent, no generation, no side effects). Body {author, content, idempotency_key} + "effects"/"claimed_original_at" only when non-None (server AuthoredWriteRequest is extra="forbid"). Success 200 (replay) or 201 (fresh) → AuthoredTurnResponse dict verbatim. 404 → AuthoredHistoryUnavailable (hide-existence: feature-absent/ungranted/session-absent, indistinguishable by design — consumer falls back, never probes). Any other non-2xx → SessionApiFailed.
|
||||
PRE: [PRE-001 hard] client is not None -- assert client is not None
|
||||
PRE: [PRE-002 hard] session_id is a non-empty str -- assert session_id and isinstance(session_id, str)
|
||||
PRE: [PRE-003 hard] content is a non-empty str -- assert content and isinstance(content, str)
|
||||
PRE: [PRE-004 hard] idempotency_key is a non-empty str -- assert idempotency_key and isinstance(idempotency_key, str)
|
||||
PRE: [PRE-005 hard] author is a non-empty str -- assert author and isinstance(author, str)
|
||||
POST: [POST-001 side_effect] exactly one POST to /sessions/{session_id}/history; body == {"author": author, "content": content, "idempotency_key": idempotency_key} plus "effects" iff effects is not None plus "claimed_original_at" iff claimed_original_at is not None (no null-valued keys — extra="forbid")
|
||||
POST: [POST-002 return_value] on 200 or 201 returns resp.json() unmodified
|
||||
ERROR_ROUTING:
|
||||
HTTP 404 (hide-existence session_not_found):
|
||||
local_handling: raise AuthoredHistoryUnavailable(session_id=session_id)
|
||||
flow_control: abort
|
||||
state_recovery: caller treats as feature-absent; fall back to a model-generated greeting; NEVER capability-probe (INV-347-1)
|
||||
HTTP other non-2xx (incl. 409 generation_active, 422 content_too_long/validation_failed, 401 auth_revoked, 410 session_retired):
|
||||
local_handling: raise SessionApiFailed(status=resp.status_code, body=resp.content)
|
||||
flow_control: abort
|
||||
state_recovery: none (409 retryable; 422 caller bug/oversize)
|
||||
STEPS:
|
||||
1. [setup, flexibility=prescriptive] assert PRE-001..PRE-005
|
||||
2. [sequential, flexibility=prescriptive] body = {"author": author, "content": content, "idempotency_key": idempotency_key}; IF effects is not None: body["effects"] = effects; IF claimed_original_at is not None: body["claimed_original_at"] = claimed_original_at
|
||||
3. [sequential, flexibility=prescriptive] resp = await client.post(f"/sessions/{session_id}/history", json=body)
|
||||
tool: { destructive: false, idempotent: true, read_only: false, open_world: false }
|
||||
4. [branch, flexibility=prescriptive] IF resp.status_code in (200, 201): RETURN resp.json(); ELIF resp.status_code == 404: RAISE AuthoredHistoryUnavailable(session_id=session_id); ELSE RAISE SessionApiFailed(status=resp.status_code, body=resp.content)
|
||||
TESTS:
|
||||
happy_fresh_201 [happy,tracer]: 201 {author:"assistant", seq:0, phase:"seeded", turn_id, content_chars, session_id, injected_at} → dict verbatim; outbound body == {"author":"assistant","content":<c>,"idempotency_key":<k>} exactly (no effects/claimed_original_at keys)
|
||||
happy_replay_200 [happy]: 200 (same-key replay, byte-identical body) → dict verbatim
|
||||
body_includes_effects [trace]: effects="none" → outbound body has "effects":"none"; claimed_original_at="2020-01-01T00:00:00Z" → body has that key too
|
||||
hide_existence_404 [error]: 404 {error_code:"session_not_found"} → raises AuthoredHistoryUnavailable(session_id=<arg>), NOT SessionApiFailed
|
||||
generation_active_409 [error]: 409 {error_code:"generation_active"} → SessionApiFailed(status=409)
|
||||
content_too_long_422 [error]: 422 {error_code:"content_too_long"} → SessionApiFailed(status=422)
|
||||
empty_content [adversarial]: content="" → AssertionError; no HTTP issued
|
||||
empty_idempotency_key [adversarial]: idempotency_key="" → AssertionError; no HTTP issued
|
||||
empty_session_id [adversarial]: session_id="" → AssertionError; no HTTP issued
|
||||
|
||||
FN get_session_messages(client: httpx.AsyncClient, session_id: str) -> dict[str, Any]
|
||||
BRIEF: GET /sessions/{session_id}/messages — the session's message history (spec §GET /sessions/{id}/messages), un-deferred as the #347 probe's read-back so a seeded turn can be confirmed to render as a normal role=assistant message (model-invisible provenance — a seed is indistinguishable from a lived turn on read). Returns {session_id, items:[{seq, role, content, ...}], next_cursor} verbatim. Owner-scoped; any non-200 → SessionApiFailed. v1 reads the server default page (no pagination params — the probe reads a fresh 1-message session; add limit/cursor when a caller needs scrollback).
|
||||
PRE: [PRE-001 hard] client is not None -- assert client is not None
|
||||
PRE: [PRE-002 hard] session_id is a non-empty str -- assert session_id and isinstance(session_id, str)
|
||||
POST: [POST-001 return_value] on 200 returns resp.json() unmodified
|
||||
ERROR_ROUTING:
|
||||
HTTP non-200 (incl. 404 session_not_found cross-owner/unknown):
|
||||
local_handling: raise SessionApiFailed(status=resp.status_code, body=resp.content)
|
||||
flow_control: abort
|
||||
state_recovery: none
|
||||
STEPS:
|
||||
1. [setup, flexibility=prescriptive] assert PRE-001, PRE-002
|
||||
2. [sequential, flexibility=prescriptive] resp = await client.get(f"/sessions/{session_id}/messages")
|
||||
3. [branch, flexibility=prescriptive] IF resp.status_code == 200: RETURN resp.json(); ELSE RAISE SessionApiFailed
|
||||
TESTS:
|
||||
happy [happy]: 200 {session_id, items:[{seq:0, role:"assistant", content:"…"}], next_cursor:null} → dict verbatim
|
||||
not_found_404 [error]: 404 → SessionApiFailed(status=404)
|
||||
empty_session_id [adversarial]: "" → AssertionError; no HTTP issued
|
||||
```
|
||||
|
||||
@@ -0,0 +1,198 @@
|
||||
---
|
||||
contract_version: "2.1"
|
||||
module: "ratatoskr.web"
|
||||
purpose: "v0.19.2 web debug-surface parity: 3 admin/debug panes (Tools inventory, BifrostState, AdminEvents SSE) proxied server-side with the admin key server-held, plus a reliable PAD-refresh poll and a non-engine reasoning indicator in the transcript."
|
||||
target_module: "ratatoskr.web (server.py routes + entrypoint.py + static/index.html)"
|
||||
scope: "v0.19.2 web debug-surface parity — bring the browser surface (now the PRIMARY debug surface) to TUI parity. THREE new admin/debug panes proxied server-side + TWO transcript affordances. (1) Tools inventory: GET /api/sessions/{id}/tools proxies owner-scoped get_session_tools into the tools pane (what the LLM HAS at turn-fire), above the live tool events. (2) BifrostState pane: GET /api/sessions/{id}/bifrost proxies admin-scoped get_session_bifrost; the admin key is SERVER-HELD (app.state.admin_key from RATATOSKR_ADMIN_API_KEY), never sent to the browser. (3) AdminEvents pane: GET /api/admin/events is an SSE proxy of stream_admin_events, session-filtered SERVER-side (heartbeats + other-session events dropped), re-emitted under a fixed 'admin_event' name so every dotted type renders with one browser listener. (4) PAD refresh: the persona/affect pane polls a bounded window instead of a single 2s shot that raced the post-turn-async affect.emit. (5) Reasoning indicator: an ephemeral, clearly-non-engine transcript line on `thinking` deltas, cleared when text begins. Direct in-session TDD (the #17/#18 pattern); this contract is authored post-implementation to anchor the heid code review (the client wrappers get_session_tools/get_session_bifrost/stream_admin_events are already contracted in the sessions/sse_client specs — this contract governs the WEB proxy + presenter surface only. v0.20.0 REDESIGN (Claude Design 'Ratatoskr Console' import): the tabbed telemetry column is replaced by a 3-column command-console — a left engine-ticker rail (the DEBUG + ADMIN + tool/turn-lifecycle feeds MERGED into one timeline via tickerAdd, plus a tools-armed chip list + a full-detail Bifrost rail pane) · a center conversation (per-turn INLINE chain-of-thought, replacing the separate Think pane) · a right resizable affect console (dominant/canonical-mood centerpiece + bipolar PAD faders each carrying a turn-to-turn Δ+sparkline + a P×A mood orbit + relations metric rows + canonical directive). ALL SERVER ROUTES UNCHANGED. Single-file/no-CDN/vanilla preserved; adds a light/dark theme toggle (dark default) + an inlined data-URI favicon. Presenter FN renames tracked below (renderBifrostState→renderBifrost; renderAffectPane→renderConsole; setPersonaStrip removed; tickerAdd/setFader/setFaderTrend/renderOrbit/renderDominant/renderDerived/renderRelations/renderDirective added). INV-001/INV-004 held.)."
|
||||
depends_on:
|
||||
- "httpx"
|
||||
- "starlette"
|
||||
- "ratatoskr.sessions" # get_session_tools, get_session_bifrost, SessionApiFailed
|
||||
- "ratatoskr.sse_client" # stream_admin_events, AdminEvent, SseConnectFailed/Dropped
|
||||
used_by:
|
||||
- "ratatoskr.web.entrypoint" # passes admin_key=RATATOSKR_ADMIN_API_KEY into create_app
|
||||
language: "python + vanilla JS (single-file SPA, no build)"
|
||||
complexity: "medium"
|
||||
estimated_loc: 290
|
||||
confidence: 0.8
|
||||
assumptions:
|
||||
- "The three client wrappers exist and are already contracted: get_session_tools(client, session_id)->dict (owner-scoped, consumer bearer; non-200 -> SessionApiFailed), get_session_bifrost(client, session_id, *, admin_key)->dict (OVERRIDES Authorization with admin_key; non-200 -> SessionApiFailed), stream_admin_events(client, *, admin_key)->AsyncIterator[AdminEvent] (non-200 -> SseConnectFailed; mid-drop -> SseConnectionDropped). The web routes are thin proxies over them; they add NO new upstream semantics."
|
||||
- "AdminEvent = {id:int, type:str, timestamp:str|None, data:dict}. data MOST carry session_id (INV-049). type is a dotted namespace (session.*/turn.*/key.*/system.*)."
|
||||
- "The web SPA is a single static/index.html served per-request via FileResponse (edits land on browser refresh; server code changes need a restart). Model/tool/admin content is UNTRUSTED text (INV-004) — every render path escapes first (esc() via textContent, or JSON.stringify wrapped in esc())."
|
||||
- "The internal-LAN trust model (0.0.0.0, no auth/TLS/CORS) is deliberate operator direction. Admin-scoped DATA becoming LAN-visible is accepted under that model; the admin KEY must nonetheless never cross to the browser."
|
||||
- "Tests: respx mocks the upstream endpoints (absolute w.example URLs) driven through the TestClient; the AdminEvents SSE proxy is tested with a finite mocked SSE byte-stream asserting the filter + fixed event name. Live-proven against ratatoskr:sindra on personal :8081."
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
functions:
|
||||
- name: "_session_tools_endpoint"
|
||||
signature: "async _session_tools_endpoint(request: Request) -> JSONResponse"
|
||||
description: "GET /api/sessions/{session_id}/tools — proxy owner-scoped tool inventory."
|
||||
preconditions:
|
||||
- "session_id in path_params."
|
||||
postconditions:
|
||||
- "POST-001: 200 with the upstream inventory dict verbatim on success."
|
||||
- "POST-002: on SessionApiFailed(status) -> JSONResponse({error_code:'session_tools_unavailable', status}, status_code=status) — status-preserving."
|
||||
steps: "Open client_factory() client; await get_session_tools(client, session_id); return 200. Except SessionApiFailed -> status-preserving envelope."
|
||||
flexibility: "prescriptive"
|
||||
|
||||
- name: "_session_messages_endpoint"
|
||||
signature: "async _session_messages_endpoint(request: Request) -> JSONResponse"
|
||||
description: "GET /api/sessions/{session_id}/messages — proxy the session's message history so the SPA renders existing turns on open (notably a #347 authored first-message seeded at create-time; without it a seeded session's transcript is blank until the user speaks)."
|
||||
preconditions:
|
||||
- "session_id in path_params."
|
||||
postconditions:
|
||||
- "POST-001: 200 with the upstream {session_id, items, next_cursor} dict verbatim on success."
|
||||
- "POST-002: on SessionApiFailed(status) -> JSONResponse({error_code:'session_messages_unavailable', status}, status_code=status) — status-preserving."
|
||||
steps: "Open client_factory() client; await get_session_messages(client, session_id); return 200. Except SessionApiFailed -> status-preserving envelope."
|
||||
flexibility: "prescriptive"
|
||||
|
||||
- name: "_session_bifrost_endpoint"
|
||||
signature: "async _session_bifrost_endpoint(request: Request) -> JSONResponse"
|
||||
description: "GET /api/sessions/{session_id}/bifrost — proxy admin-scoped Bifrost dispatch state."
|
||||
preconditions:
|
||||
- "session_id in path_params."
|
||||
- "PRE-001 (fail-visible): app.state.admin_key must be truthy; else 400 admin_key_not_configured with NO upstream call."
|
||||
postconditions:
|
||||
- "POST-001: the admin key is read from app.state.admin_key ONLY; it is passed to get_session_bifrost(admin_key=...) and NEVER placed in a response body or surfaced to the browser."
|
||||
- "POST-002: 200 with the upstream state dict verbatim on success."
|
||||
- "POST-003: on SessionApiFailed(status) -> {error_code:'bifrost_state_unavailable', status} at status_code=status (notably 404 not-bound, 403 scope-denied)."
|
||||
steps: "If not admin_key -> 400. Open client; await get_session_bifrost(client, session_id, admin_key=admin_key); 200. Except SessionApiFailed -> status-preserving envelope."
|
||||
flexibility: "prescriptive"
|
||||
|
||||
- name: "_admin_event_matches_web"
|
||||
signature: "_admin_event_matches_web(ev: AdminEvent, session_id: str | None) -> bool"
|
||||
description: "AdminEvents session-filter (mirrors the TUI _admin_event_matches, design-brief §6)."
|
||||
postconditions:
|
||||
- "POST-001: ev.type == 'system.heartbeat' -> False (keepalive noise dropped)."
|
||||
- "POST-002: ev.type.startswith('system.') (non-heartbeat) -> True (stream-integrity signals always pass)."
|
||||
- "POST-003: otherwise -> True IFF session_id is not None AND ev.data.get('session_id') == session_id (per-session scoping; a None session_id forwards NO non-system event)."
|
||||
flexibility: "prescriptive"
|
||||
|
||||
- name: "_admin_events_endpoint"
|
||||
signature: "async _admin_events_endpoint(request: Request) -> Response"
|
||||
description: "GET /api/admin/events?session_id=... — SSE proxy of stream_admin_events, session-filtered server-side."
|
||||
preconditions:
|
||||
- "PRE-001 (fail-visible): app.state.admin_key truthy; else 400 admin_key_not_configured with NO stream opened."
|
||||
postconditions:
|
||||
- "POST-001: returns StreamingResponse(media_type='text/event-stream'); the admin key never crosses to the browser."
|
||||
- "POST-002: ONLY events passing _admin_event_matches_web(ev, session_id) are forwarded; each is re-emitted under the FIXED SSE event name 'admin_event' with {id,type,timestamp,data} in the payload (the real dotted type rides in the payload, so one browser listener renders every type — nothing silently dropped by name)."
|
||||
- "POST-003: SseConnectFailed/SseConnectionDropped/MalformedSseId/MalformedSseData -> a single 'stream_error' SSE frame, then the stream ends (best-effort; never raises to the browser)."
|
||||
- "POST-004: asyncio.CancelledError (browser disconnect) re-raises to unwind the generator; the upstream client is aclose()'d in finally on every exit path."
|
||||
steps: "If not admin_key -> 400. gen(): open client; async-for ev in stream_admin_events(admin_key); skip unless _admin_event_matches_web; yield _format_sse('admin_event', {...}). Except SSE errors -> yield stream_error. Except CancelledError -> raise. Finally aclose(). Return StreamingResponse(gen())."
|
||||
flexibility: "prescriptive"
|
||||
|
||||
- name: "create_app (amendment)"
|
||||
signature: "create_app(client_factory, *, end_user_id=None, bifrost_consumer_key=None, bifrost_visible_host=None, affect_read_url=None, memory_read_url=None, admin_key=None) -> Starlette"
|
||||
description: "New optional admin_key param stored at app.state.admin_key; entrypoint passes RATATOSKR_ADMIN_API_KEY. v0.20.7 adds memory_read_url (app.state.memory_read_url, from RATATOSKR_MEMORY_READ_URL) for the memory viewer. Four new routes registered across the arc."
|
||||
postconditions:
|
||||
- "POST-001: app.state.admin_key = admin_key (default None -> the two admin routes fail-visible per their PRE-001)."
|
||||
- "POST-002: routes /api/sessions/{session_id}/tools, /api/sessions/{session_id}/bifrost, /api/admin/events added; existing routes unchanged."
|
||||
- "POST-003 (v0.20.7): app.state.memory_read_url = memory_read_url; route /api/memory/chunks added (the memory-viewer proxy)."
|
||||
flexibility: "closed"
|
||||
|
||||
- name: "reasoning indicator (index.html: showThinkingNote / hideThinkingNote)"
|
||||
signature: "showThinkingNote() ; hideThinkingNote() // called from the turn SSE loop"
|
||||
description: "Ephemeral transcript affordance signalling reasoning inference — clearly NOT engine output."
|
||||
postconditions:
|
||||
- "POST-001: on the first `thinking` delta, an italic '<Agent> <phrase>' line (✦ glyph, rotating phrase) is shown; it supersedes any live 'awaiting first token' heartbeat."
|
||||
- "POST-002: the agent display name is derived from state.agentId and rendered via textContent (NEVER innerHTML) — INV-004 holds even for an adversarial agent_id."
|
||||
- "POST-003: it is removed the instant the first `text` delta arrives, and on any terminal (done/error/cancelled); the rotation interval is cleared on removal (no leaked setInterval)."
|
||||
flexibility: "prescriptive"
|
||||
|
||||
- name: "PAD refresh poll (index.html: terminal() done-branch)"
|
||||
signature: "on Done: poll loadPersona over [1500,3500,6500,10500]ms"
|
||||
description: "Catch the post-turn-async affect.emit without racing it (replaces the single 2s shot)."
|
||||
postconditions:
|
||||
- "POST-001: loadAffect sets state.lastAffectAt = snap.emitted_at; the poll captures beforeAt and stops (settled) once state.lastAffectAt !== beforeAt."
|
||||
- "POST-002: a scheduled poll no-ops if a NEW turn has started (state.turnId truthy) or already settled — no refresh of a stale agent, no unbounded polling."
|
||||
flexibility: "open"
|
||||
|
||||
- name: "loadTranscript (index.html)"
|
||||
signature: "async loadTranscript(sessionId) -> void"
|
||||
description: "On session open, GET /api/sessions/{id}/messages and render each EXISTING turn into #transcript — notably a #347 authored first-message seeded at create-time (which lives in the ledger, not the live turn stream, so without this the transcript is blank until the user speaks)."
|
||||
postconditions:
|
||||
- "POST-001: assistant items render as a .response .md-body bubble via markdownSafe(content) (escape-first whitelist, same path as appendResponse); user items render as a .prompt-echo via textContent — no upstream content reaches innerHTML unescaped (INV-004)."
|
||||
- "POST-002: any non-200, fetch error, or parse error is swallowed (best-effort) — a blank transcript is acceptable; opening the workspace is never blocked."
|
||||
flexibility: "prescriptive"
|
||||
|
||||
- name: "web pane renderers (v0.20.0: index.html: renderToolsInventory / renderBifrost / openAdminEvents + tickerAdd)"
|
||||
signature: "renderToolsInventory(inv) ; renderBifrost(b) ; openAdminEvents(sessionId) ; tickerAdd(kind, msg, dim)"
|
||||
description: "Render the debug/admin surfaces into the 3-column console; all content escaped (INV-004). v0.20.0: BifrostState is now a full-detail LEFT-RAIL pane (renderBifrost, renamed from renderBifrostState); AdminEvents + the raw debug/op log + tool_start/result + turn lifecycle are MERGED into one engine-ticker timeline via tickerAdd (openAdminEvents routes admin_event → tickerAdd; the turn SSE handlers route worker_phase/tool/text_boundary/affect_update → tickerAdd); Tools inventory is a rail chip list (renderToolsInventory)."
|
||||
postconditions:
|
||||
- "POST-001: every dynamic value (agent_id, tool names/descriptions, endpoint, caps, consumer_id, admin event type + data, ticker msg/dim) is passed through esc() or esc(JSON.stringify(...)); no upstream string reaches innerHTML unescaped."
|
||||
- "POST-002: openAdminEvents closes a prior EventSource before opening a new one (state.adminES) and, on stream_error, closes so native EventSource does NOT retry-loop; admin events render into the engine ticker via tickerAdd."
|
||||
- "POST-003: renderToolsInventory renders builtin + bifrost tool NAMES as rail chips (a compact 'what does the LLM have' glance); renderBifrost renders endpoint + connected + consumer_id + capabilities_granted chips + per-tool name/description rows (the full detail, admin-gated; the admin key stays server-held). tickerAdd bounds the feed to the last 400 rows (a tail, not an archive)."
|
||||
- "POST-004: tool NAME-vs-DESCRIPTION split preserved — the rail chip list shows names only; per-tool descriptions live in the Bifrost pane's tools list. The engine-ticker spine (.ticker-inner::before) lives on the content-height wrapper so it stays visible when auto-scrolled to the newest entry."
|
||||
flexibility: "open"
|
||||
|
||||
- name: "renderConsole + trend (v0.20.0 — unified persona/affect console; supersedes renderAffectPane/renderPersonaPane/setPersonaStrip)"
|
||||
signature: "renderConsole(snap) ; setFader(axis,v) ; setFaderTrend(axis) ; renderOrbit() ; renderDominant(snap) ; renderDerived(snap) ; renderRelations(snap) ; renderDirective(snap) ; pushAffectHistory(snap) ; sparkPointsH(vals,w,h,endX) ; padDeltas(vals) ; deltaStrip(deltas) ; orbitFrame(H,head,ts) ; orbitProj/orbitShadowY/orbitWallPt/orbitAxisPt ; startOrbitAnim() ; trendDelta(vals)"
|
||||
description: "ONE render path for BOTH the Tier-1 persona_state snapshot and the Tier-3 affect snapshot (renderConsole), feeding the right affect console: dominant/canonical-mood centerpiece, bipolar PAD faders (each with a turn-to-turn Δ + sparkline), a P×A mood orbit from PAD history, an affect-derived grid, relations metric rows, and the canonical directive. Replaces the v0.19.x split of renderPersonaPane (Tier-1 pane) + renderAffectPane (Tier-3 pane) + setPersonaStrip (top-bar strip, removed — PAD now lives in the console faders)."
|
||||
postconditions:
|
||||
- "POST-001: reads snap.relations (relation_edge/1: target_entity + trust_ability/benevolence/integrity + warmth as {value,confidence,evidence_count} + agency + relation_context) — the CURRENT Worldtree emit shape; falls back to the legacy flat snap.valence for an older emitter. Tier-1 fields (baseline_pad, mood_drift, dominant_emotion, emotions_active) render WHEN PRESENT, '—' when absent (Tier-3 lacks them)."
|
||||
- "POST-002: SVG sparklines + affect visuals (v0.20.4, adapted from the design prototype; v0.20.7 = design iteration-3). Each relation metric shows a HORIZONTAL SVG sparkline (`sparkPointsH`, 56×13, auto-scaled to its OWN range, sparkFade gradient + end dot), now BACKED by a subtle grid (`<pattern id=sparkGrid>` + a bg `<rect>` behind the polyline). Each PAD fader shows current value + Δ-vs-previous (▲/▼) + a per-turn Δ STRIP: v0.20.7 REPLACES the vertical polyline strip (removed `stripPoints`) with a column of 12 diverging HTML bars (`padDeltas`→`deltaStrip`, newest at bottom, each bar offset L/R of a center line by that turn's Δ, magnitude→width, age→opacity, zero-Δ→faint center dot). renderOrbit is now a DIMETRIC OPEN BOX (azimuth 35° / elevation 25°, D→right / A→left-back / P→up; removed the isometric `proj3` for `orbitProj/orbitShadowY/orbitWallPt/orbitAxisPt`) — a ghost A×P wall (carrying the P readout) + a D×A floor, JS-DRIVEN animated replay (`orbitFrame` rebuilt per rAF frame by a singleton `startOrbitAnim` loop reading live `ORBIT_HIST`; no SMIL/CSS-keyframes; reduced-motion → static final-state render). All drawn from AFFECT_HIST (rolling, HIST_CAP=24, session-lived); coords are computed numerics (no upstream strings → INV-004 trivially held). Gradients/patterns live in one hidden `<defs>` svg in the console. v0.20.9 (R32-1B prep): the fader fill (`padFillFrac`) + orbit projections (`_padNorm`) AUTO-SCALE to the session's own max |PAD| (`padScale`, floor 1.0) instead of hard-clamping to [-1,1] — so an unbounded-z PAD (Worldtree R32-1B) renders at FULL range and never pegs/escapes the frame, while today's [-1,1] values are unchanged (scale==1); the exact value is always shown numerically (unclamped). This scaling is PURELY debug-display — ratatoskr is a downstream observer; it never touches the agent's real affect or any write path (the `--set-persona-pad` seed carries values unclamped)."
|
||||
- "POST-003: pushAffectHistory dedupes by emitted_at||last_updated_at so the ~4x/turn post-turn PAD poll contributes ONE sample/turn; history is CLIENT-side only (lost on reload — durable cross-session history via a provider-side snapshot log is a deferred follow-up, NOT built here)."
|
||||
- "POST-004: INV-001 honesty — no fabricated Tier-1 fields. The dominant-emotion centerpiece shows a real OCC dominant_emotion (Tier-1) OR the CANONICAL mood word from canonMood(pad) (Tier-3, dimmed) OR '—'; NEVER a synthesized emotion. The affect-derived grid drops non-emitted metrics (intensity/decay-τ) and shows only real/client-derived cells (baseline/drift real for Tier-1, client-derived samples/volatility). INV-004 — every dynamic value passes through esc(); numerics go through toFixed, never innerHTML-raw."
|
||||
flexibility: "open"
|
||||
|
||||
- name: "canonical affect-NL + context-injection reconstruction (v0.19.5 canons; v0.20.2 full context-injection panel)"
|
||||
signature: "canonMood(pad) ; canonDirective(rel) ; canonPadFallback(pad) ; canonEmotionDirective(type) ; renderDirective(snap) ; loadPersonaCanon()"
|
||||
description: "Reconstruct + render the HIDDEN affect-context block Worldtree assembles into the agent's system prompt (never on any wire) — byte-exact to Worldtree's own describe_pad + render_d2_canonical + derive_directive + _pad_band_fallback. The v0.20.2 'context injection' panel shows the full block: mood descriptor + mood directive + relationship directive. Reference: docs/vendor/worldtree-persona-canon/affect-egress-consumer-reference.md (pinned)."
|
||||
postconditions:
|
||||
- "POST-001: DETERMINISTIC, no LLM. canonMood mirrors describe_pad (valence×arousal grid + strict ±0.3 bands + dominance clause); canonDirective mirrors render_d2_canonical; canonPadFallback mirrors renderer._pad_band_fallback BYTE-EXACT (P×A quadrant: hi/lo/mid arousal band × p>0.3/<-0.3/neutral, with the negative_low_dominance (d<-0.3) special case + neutral_high_a + default); canonEmotionDirective is the occ_directives[type].directive lookup (+ tier / full_only flag)."
|
||||
- "POST-002: the canon DATA is VENDORED (docs/vendor/worldtree-persona-canon/{d2-mood-render-canon-v1,d2-render-canon-v1}.json), pinned drift-gated in .corviduo-canonicals.toml; the flat browser form (static/persona_render_canon.json, served /static) is regenerated by scripts/build_persona_canon.py via Worldtree's OWN authoritative loader — v0.20.2 extended it to emit mood_directive {occ_directives, pad_band_fallback, salience, pad_band_cutoff, full_only}. The affect-egress consumer reference is pinned tolerate_drift (worldtree-affect-egress-consumer-reference-v1; worldtree-dev co-signs + pings on change)."
|
||||
- "POST-003: fail-open — canon absent (fetch fails) → the reconstructed lines OMIT, the structured console still renders. Every canon-derived string is esc()'d before the DOM (INV-004)."
|
||||
- "POST-004: HONEST-PARTIAL provenance (affect-egress-reference §3). The mood descriptor + relationship directive are EXACT (tagged 'exact'); the mood DIRECTIVE is a CANDIDATE pair (tagged 'candidate') — the OCC emotion directive for the delivered dominant_emotion type AND the PAD-band fallback — because affect.emit is type-only (no intensity) so the salience gate (≥0.2) can't be evaluated; BOTH are shown with the 'injected if intensity ≥ salience' caveat, never asserting which fires. When dominant_emotion is absent the fallback alone is EXACT. The panel is labeled reconstructed + hidden-from-consumers + dev-only (the reference-impl's sanctioned understand/reconstruct use, NOT end-user display per the reference's caveat). WATCH: a pending Worldtree render_d2_canonical change conditionally drops the trailing 'avoid premature we-framing' clause under a 3-gate combo — canonDirective holds as-is until worldtree-dev pings with the exact conditional + a canon bump."
|
||||
flexibility: "open"
|
||||
|
||||
- name: "memory viewer (v0.20.7 — provider debug read → web proxy → console pane)"
|
||||
signature: "server: _memory_chunks_endpoint(request) [GET /api/memory/chunks] ; provider: add_memory_read_route(app, store) [GET /memory/chunks] + RatatoskrMemoryStore.list_chunks(*, agent_id, end_user_id) + .count_chunks() ; index.html: loadMemory(agentId) ; renderMemory(data) ; setMemHead(count, total)"
|
||||
description: "Durable memory chunks Worldtree promoted into OUR store, surfaced as a live-polling MEMORY console pane (content·scope·origin·revision per chunk). Mirrors the #18-D2 affect read pattern: a NON-bifrost debug read on OUR own store (bifrost's memory protocol has no list-all verb) → a web proxy supplying end_user_id server-side → the pane. Polled on session open + the post-turn window (promotion is async, like affect.emit)."
|
||||
postconditions:
|
||||
- "POST-001 (provider read): GET /memory/chunks?agent_id=&end_user_id= returns {chunks:[{chunk_id,content,scope,origin,revision}], count, total}. end_user_id REQUIRED (400 missing_end_user_id) — the partition boundary. Filter: end_user STRICT (scope.end_user==end_user_id), agent_id LENIENT (excluded only if the chunk CARRIES an agent_self axis that differs — so an {end_user}-only chunk, the real WT promotion shape, is not hidden). An empty match is a 200 empty list (0-chunks is a visible answer, never a 404). `total` = unfiltered store-wide count (distinguishes empty-store from scope-mismatch). content = best-effort text field / distillate summary / compact JSON-minus-embedding — a DEBUG read; bifrost verbs stay index/conduit-faithful."
|
||||
- "POST-002 (web proxy): GET /api/memory/chunks supplies end_user_id from app.state.end_user_id (NEVER the browser), forwards the browser-named agent_id, proxies to app.state.memory_read_url (the combined :8392 provider serves both read routes). 400 memory_not_configured when unset; 502 memory_provider_unreachable on network error; status passthrough otherwise. Mirrors _affect_state_endpoint (#18 D2 INV-002)."
|
||||
- "POST-003 (pane): renderMemory shows count(matched)/total(store-wide) in the head + one .mem-chunk per chunk (scope axes + origin + revision + content, ALL esc()'d — INV-004). Empty states are honest + diagnostic: total 0 → 'no memory chunks yet — promotion needs a bound memory/combined session + ~6 turns (or idle); if 0/0 the bind wasn't memory-granted or closed pre-promotion'; total>0 → 'scope mismatch, not an empty store'."
|
||||
flexibility: "open"
|
||||
|
||||
- name: "markdownSafe pass-2 (v0.20.6 RP coloring + v0.20.7 tables / nested lists / streaming)"
|
||||
signature: "markdownSafe(raw) ; mdTable(lines, i) ; mdInline(s)"
|
||||
description: "The escape-first whitelist Markdown renderer, extended pass-2: GFM pipe tables, indentation-nested lists, ordered-list start numbering, and streaming-partial robustness. Pass-1 (RP speech/action coloring + CommonMark paragraph reflow) shipped v0.20.6."
|
||||
postconditions:
|
||||
- "POST-001: GFM pipe tables (`mdTable`) — a pipe row + an alignment/delimiter row (`|---|:--:|`) → <table class=md-table> with per-column text-align from the delimiter colons; body rows parsed until a non-pipe line."
|
||||
- "POST-002: indentation-nested lists — leading-space depth builds a stack of <ul>/<ol> with each child list INSIDE the open parent <li> (valid nested HTML); same-level items are siblings; ul↔ol switches close+reopen. Ordered lists honor the first item's number (<ol start=N> when != 1)."
|
||||
- "POST-003: streaming robustness — an unterminated code fence renders as a partial code block; a table header without its delimiter yet falls through to a paragraph (becomes a table once the delimiter streams in); parsing never throws on a partial. INV-004 held — esc() runs FIRST on the whole input, so table cells / list items / code all carry escaped content."
|
||||
flexibility: "open"
|
||||
|
||||
invariants:
|
||||
- "INV-004 (untrusted-render): ALL model / tool / admin / agent-supplied text is escaped before entering the DOM (esc via textContent, or esc(JSON.stringify)). No new render path introduces an innerHTML sink for upstream content. This is the highest-value review target — the new JS render paths are NOT unit-tested. v0.20.7: the memory pane (chunk content/scope/origin), the delta-strip bars, and markdownSafe table cells / list items all pass through esc() (esc runs FIRST on the whole markdown input)."
|
||||
- "INV-ADMIN-KEY: the admin key exists ONLY at app.state.admin_key (from RATATOSKR_ADMIN_API_KEY). It is never serialized into any response, never sent to the browser, never logged. The browser receives only the session-filtered RESULT of admin-scoped reads."
|
||||
- "INV-FILTER: AdminEvents filtering happens SERVER-side (_admin_event_matches_web) — the browser never receives the cross-session admin firehose; only active-session events + non-heartbeat system.* cross the wire."
|
||||
- "INV-FAIL-VISIBLE: both admin routes return 400 admin_key_not_configured when the key is absent — never a silent empty pane, never an upstream call with an empty bearer."
|
||||
- "INV-LIFECYCLE: SSE generators and EventSources are cleaned up on every exit path (upstream client aclose() in finally; setInterval cleared in hideThinkingNote; prior EventSource closed before re-open) — no leaked connections, tasks, or timers."
|
||||
- "INV-ADDITIVE: existing routes, panes, and the turn-stream path are unchanged; the 3 new routes + 2 new tabs are purely additive (59 web tests incl. all prior ones stay green)."
|
||||
---
|
||||
|
||||
# v0.19.2 — web debug-surface parity (BifrostState · AdminEvents · Tools · PAD-poll · reasoning)
|
||||
|
||||
## Context
|
||||
|
||||
The browser surface is now the operator's PRIMARY debug surface, and it lagged the
|
||||
TUI: the TUI gained Tools/BifrostState/AdminEvents panes (v0.18.9–.11) that were never
|
||||
ported to the web. This change closes that gap and adds two transcript affordances (a
|
||||
reliable PAD refresh + a reasoning indicator). The client wrappers already existed and
|
||||
are contracted elsewhere; this contract governs the WEB proxy routes + the SPA presenter
|
||||
paths, whose JS render code is not unit-tested — hence the cross-frontier code review.
|
||||
|
||||
## Review focus (for the heid panel)
|
||||
|
||||
1. **INV-004 escaping** in every new render path — the un-unit-tested surface; the exact
|
||||
class of bug (`renderPersonaPane` fabricating a Tier-1 field) that only a cross-model
|
||||
review caught on #18 D2.
|
||||
2. **INV-ADMIN-KEY** — confirm the admin key never reaches a response body or the browser.
|
||||
3. **AdminEvents SSE proxy** (`_admin_events_endpoint`) — generator/filter/lifecycle: fixed
|
||||
event name, server-side filter, `stream_error` on failure, `aclose()` on every path,
|
||||
`CancelledError` re-raise on disconnect.
|
||||
4. **PAD-poll** stop-condition — does `emitted_at` advancement + the `state.turnId` guard
|
||||
correctly stop the poll without racing or leaking timers?
|
||||
5. **Reasoning indicator** lifecycle — shown on first `thinking`, removed on first `text`
|
||||
or terminal, interval cleared (no leaked `setInterval`), name via `textContent`.
|
||||
@@ -1,6 +1,50 @@
|
||||
{
|
||||
"components": {
|
||||
"schemas": {
|
||||
"AuthoredTurnResponse": {
|
||||
"description": "#347 — the authored-write ack. Provenance is audit-only and NEVER on this\npayload (INV-347-7). ``content_chars`` is the CHARACTER count (may differ from\nthe UTF-8 byte length the request is bounded against — INV-347-12).",
|
||||
"properties": {
|
||||
"author": {
|
||||
"title": "Author",
|
||||
"type": "string"
|
||||
},
|
||||
"content_chars": {
|
||||
"title": "Content Chars",
|
||||
"type": "integer"
|
||||
},
|
||||
"injected_at": {
|
||||
"title": "Injected At",
|
||||
"type": "string"
|
||||
},
|
||||
"phase": {
|
||||
"title": "Phase",
|
||||
"type": "string"
|
||||
},
|
||||
"seq": {
|
||||
"title": "Seq",
|
||||
"type": "integer"
|
||||
},
|
||||
"session_id": {
|
||||
"title": "Session Id",
|
||||
"type": "string"
|
||||
},
|
||||
"turn_id": {
|
||||
"title": "Turn Id",
|
||||
"type": "integer"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"turn_id",
|
||||
"session_id",
|
||||
"author",
|
||||
"phase",
|
||||
"seq",
|
||||
"injected_at",
|
||||
"content_chars"
|
||||
],
|
||||
"title": "AuthoredTurnResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"BifrostBindingRequest": {
|
||||
"additionalProperties": false,
|
||||
"description": "Bifrost binding parameters for session-create (issue #160).",
|
||||
@@ -390,6 +434,7 @@
|
||||
"session_not_found",
|
||||
"session_not_bifrost_bound",
|
||||
"session_retired",
|
||||
"generation_active",
|
||||
"agent_not_available",
|
||||
"turn_not_found",
|
||||
"turn_finished",
|
||||
@@ -905,7 +950,7 @@
|
||||
"info": {
|
||||
"description": "Multi-turn conversation interface for Worldtree agents.",
|
||||
"title": "Worldtree Conversation API",
|
||||
"version": "2.2.0"
|
||||
"version": "2.3.0"
|
||||
},
|
||||
"openapi": "3.1.0",
|
||||
"paths": {
|
||||
@@ -6382,6 +6427,141 @@
|
||||
"summary": "Update Session"
|
||||
}
|
||||
},
|
||||
"/sessions/{session_id}/history": {
|
||||
"post": {
|
||||
"description": "#347 — write one model-visible turn into a session's ledger AS the bound\nagent, WITHOUT a generation and WITHOUT lived-turn side effects.\n\nHide-existence ordering (INV-347-13): the ``session.history.write`` grant is\nchecked FIRST — before session resolution, before ANY body parse/validation,\nbefore the active-generation guard. An ungranted caller receives ONLY the\nhide-404 (byte-identical to session-not-found, INV-347-1) — never a\n422/409/403 that would distinguish feature-absent from session-absent.",
|
||||
"operationId": "authored_history_write_sessions__session_id__history_post",
|
||||
"parameters": [
|
||||
{
|
||||
"in": "path",
|
||||
"name": "session_id",
|
||||
"required": true,
|
||||
"schema": {
|
||||
"title": "Session Id",
|
||||
"type": "string"
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
"201": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/AuthoredTurnResponse"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Successful Response"
|
||||
},
|
||||
"400": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Error — unified envelope (detail.error_code is the stable identifier)."
|
||||
},
|
||||
"401": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Error — unified envelope (detail.error_code is the stable identifier)."
|
||||
},
|
||||
"403": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Error — unified envelope (detail.error_code is the stable identifier)."
|
||||
},
|
||||
"404": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Error — unified envelope (detail.error_code is the stable identifier)."
|
||||
},
|
||||
"405": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Error — unified envelope (detail.error_code is the stable identifier)."
|
||||
},
|
||||
"409": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Error — unified envelope (detail.error_code is the stable identifier)."
|
||||
},
|
||||
"412": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Error — unified envelope (detail.error_code is the stable identifier)."
|
||||
},
|
||||
"422": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Error — unified envelope (detail.error_code is the stable identifier)."
|
||||
},
|
||||
"500": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Error — unified envelope (detail.error_code is the stable identifier)."
|
||||
},
|
||||
"503": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Error — unified envelope (detail.error_code is the stable identifier)."
|
||||
}
|
||||
},
|
||||
"security": [
|
||||
{
|
||||
"HTTPBearer": []
|
||||
}
|
||||
],
|
||||
"summary": "Authored History Write"
|
||||
}
|
||||
},
|
||||
"/sessions/{session_id}/messages": {
|
||||
"get": {
|
||||
"description": "Return paginated message history for a session.",
|
||||
|
||||
@@ -2699,10 +2699,13 @@ The `turn.started` event always carries `bifrost_override_applied: bool` (True/F
|
||||
|
||||
Tier 3 agents are consumer-owned, Worldtree-hosted agents whose
|
||||
identity lives at `<user_id>:<agent_name>`. They share the persistent
|
||||
session infrastructure with Tier 1 / Tier 2 but layer-specific
|
||||
machinery (persona, motivational, memory, valence) is reserved for
|
||||
later phases — Phase 2.0 ships baseline addressing + ownership +
|
||||
lifecycle only.
|
||||
session infrastructure with Tier 1 / Tier 2. The layer-specific
|
||||
machinery is now largely active: **`persona` (Phase 2.1, #186),
|
||||
`memory` (Phase 2.1, #197), and `motivational` (Phase 2.2, #187) are
|
||||
shipped and consumer-settable at define-time.** Only **`valence` remains
|
||||
deferred** (non-null → 422 `layer_deferred`). Phase 2.0 shipped the
|
||||
baseline addressing + ownership + lifecycle substrate; the subsections
|
||||
below document the active layers and their exact validated shapes.
|
||||
|
||||
### Endpoints
|
||||
|
||||
@@ -2720,11 +2723,13 @@ lifecycle only.
|
||||
{
|
||||
"agent_name": "wizard",
|
||||
"system_prompt": "You are a guided-elicitation wizard...",
|
||||
"model": "glm5-turbo",
|
||||
"persona": null, // schema-reserved; non-null → 422 layer_deferred
|
||||
"motivational": null,
|
||||
"valence": null,
|
||||
"memory": null
|
||||
"role": "gen-reasoning", // REQUIRED — a configured model-role (#344), not a raw model id
|
||||
"persona": { // active (Phase 2.1) — single-letter OCEAN keys; see "Persona layer"
|
||||
"ocean": {"O": 0.4, "C": 0.6, "E": -0.3, "A": 0.2, "N": 0.5}
|
||||
},
|
||||
"motivational": null, // active (Phase 2.2) — see "Motivational layer"
|
||||
"memory": null, // active (Phase 2.1) — see "Memory layer"
|
||||
"valence": null // still deferred — non-null → 422 layer_deferred
|
||||
}
|
||||
```
|
||||
|
||||
@@ -2756,6 +2761,152 @@ after definition.
|
||||
The 201 response includes an advisory `warnings` array (#219) — see
|
||||
"Model-assignment warnings" under `PATCH` below.
|
||||
|
||||
> **Vendoring note (OpenAPI 2.3.0).** In the frozen OpenAPI 2.3.0 document
|
||||
> the `persona` / `motivational` / `memory` / `valence` request fields
|
||||
> serialize as **untyped/freeform** — the `POST /agents/define` request
|
||||
> model types them as `Any` so the layers can activate without a
|
||||
> schema-breaking change. The shapes documented in the subsections below
|
||||
> are the **authoritative, validator-enforced** schemas; generate client
|
||||
> types from this section, not from the freeform OpenAPI fields.
|
||||
|
||||
##### Persona layer (Phase 2.1, #186)
|
||||
|
||||
`persona` is **active** as of Phase 2.1. It carries the agent's OCEAN
|
||||
personality vector — the durable trait profile from which Worldtree
|
||||
derives the mood setpoint (`baseline_pad`) and the mood dynamics
|
||||
(gain + relaxation time-constants). Shape:
|
||||
|
||||
```json
|
||||
"persona": {
|
||||
"ocean": { // REQUIRED — exactly these 5 keys, no more, no fewer
|
||||
"O": 0.4, // Openness — float in [-1.0, 1.0]
|
||||
"C": 0.6, // Conscientiousness
|
||||
"E": -0.3, // Extraversion
|
||||
"A": 0.2, // Agreeableness
|
||||
"N": 0.5 // Neuroticism
|
||||
},
|
||||
"behavioral_notes": "...", // optional, ≤ 4096 chars
|
||||
"temperament_notes": "..." // optional, ≤ 4096 chars
|
||||
}
|
||||
```
|
||||
|
||||
**⚠ OCEAN key format — single-letter, uppercase.** The `/agents/define`
|
||||
persona validator requires the `ocean` map to contain **exactly** the five
|
||||
uppercase single-letter keys `O, C, E, A, N`. This is a deliberate,
|
||||
load-bearing contrast with the transient-character primitive
|
||||
(`POST /characters`), whose `ocean` block uses the **spelled-out**
|
||||
lowercase keys (`openness`, `conscientiousness`, …). Sending spelled-out
|
||||
keys to `/agents/define` returns 422 `persona_ocean_required` ("must
|
||||
contain exactly the 5 keys O, C, E, A, N").
|
||||
|
||||
> **Fixed in v1.0.0b21 (#348).** Before that build a correctly
|
||||
> single-letter-keyed persona was accepted and stored, but resolved to a
|
||||
> **neutral** mood, because Worldtree's internal mood-derivation read the
|
||||
> spelled-out key form. On v1.0.0b21+ an API-declared persona correctly
|
||||
> drives the derived mood setpoint. If you observe neutral mood on a
|
||||
> persona-defined agent, confirm the deployment is ≥ v1.0.0b21.
|
||||
|
||||
**Range.** Each value is a float in `[-1.0, 1.0]` **signed** — `0.0` is the
|
||||
population mean, NOT `[0.0, 1.0]`. Booleans are rejected. Out-of-range → 422
|
||||
`persona_ocean_out_of_range`. See [`docs/ocean-traits.md`](ocean-traits.md)
|
||||
for the SOTA-grounded 5-band behavioural mapping.
|
||||
|
||||
Semantics:
|
||||
|
||||
- **Per-agent identity trait** — identical for every end-user and session;
|
||||
immutable post-define (`PATCH {"persona": …}` → 422 `field_not_mutable`).
|
||||
To change the OCEAN profile, delete and re-define the agent.
|
||||
- **`extensions` is reserved** — the field exists but must be empty at v0.1;
|
||||
a non-empty `extensions` returns 422 `layer_deferred`.
|
||||
- **Sets the mood SETPOINT, not the current mood.** The OCEAN vector fixes
|
||||
`baseline_pad` (the PAD point the mood relaxes toward over time); the
|
||||
*current* per-session mood point is seeded separately via
|
||||
`POST /sessions/{id}/persona_state` (below).
|
||||
|
||||
Validation 422 codes: `persona_ocean_required` (missing `ocean`, or keys
|
||||
≠ {O,C,E,A,N}), `persona_ocean_out_of_range` (a value outside [-1.0, 1.0], or
|
||||
a boolean), `persona_notes_too_large` (a note > 4096 chars), `layer_deferred`
|
||||
(non-empty `extensions`), `validation_failed` (unknown top-level field).
|
||||
|
||||
##### `POST /sessions/{session_id}/persona_state` — seed the session mood point (Phase 2.1, #186/#189)
|
||||
|
||||
Session-scoped mood seed. Sets the *current* PAD mood point for one
|
||||
session's bound agent — the starting emotional state, distinct from the
|
||||
OCEAN-derived setpoint the mood relaxes toward. Works on any
|
||||
persona-enabled session (Tier 1 or Tier 3); most useful for a Tier 3
|
||||
durable-agent session that wants to start a conversation from a specific
|
||||
mood.
|
||||
|
||||
Request:
|
||||
|
||||
```json
|
||||
{
|
||||
"pad": {
|
||||
"pleasure": 0.42, // float in [-1.0, 1.0]
|
||||
"arousal": 0.25,
|
||||
"dominance": 0.33
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Response: **`204 No Content`** — no body, no audit event (a session-scoped
|
||||
runtime overlay, not a security-relevant event).
|
||||
|
||||
Semantics:
|
||||
|
||||
- **PAD-only** (#317 Option A). The body accepts exactly one key, `pad`,
|
||||
which must carry all three of `pleasure` / `arousal` / `dominance`, each a
|
||||
float in `[-1.0, 1.0]`. Any other top-level key → 422 `validation_failed`;
|
||||
a missing or malformed `pad` → 422 `persona_seed_invalid`.
|
||||
|
||||
> **✓ R32-1B (landed, v1.0.0b29):** The PAD range `[-1.0, 1.0]` relaxes to an **unbounded latent `z`** with a finite wire sanity bound (`~±10`) as of R32 Slice-1B. The JSON shape/fields/types are UNCHANGED — only the declared range/semantics change (the value becomes a latent that renders to a bounded display value). Consumers that merely store-and-return PAD need no change; consumers that validate/clamp PAD to `[-1,1]` must relax that bound. Source of truth: `docs/contracts/persona_envelope.contract.md` rev 1.7 (INV-ENV-16).
|
||||
- **Seeds the current mood POINT, not the setpoint.** The OCEAN persona
|
||||
(above) fixes the setpoint the mood relaxes toward; this endpoint sets
|
||||
where the mood *starts*. It does not alter the persona.
|
||||
- **Cross-owner sessions return 404** (existence-hiding — a session that
|
||||
isn't yours is indistinguishable from one that doesn't exist).
|
||||
- **Pull-over-push precedence (#289).** Once a session's baseline has been
|
||||
rehydrated from an `affect.fetch` (the authoritative cross-session
|
||||
source), a later SET seed is silently ignored — the fetched baseline wins.
|
||||
|
||||
There is **no** `POST /agents/{id}/persona_state` — mood is per-session, not
|
||||
a durable agent property. `GET /agents/{agent_id}/persona_state`
|
||||
short-circuits to 404 for Tier-3 colon-ids: Tier-3 mood is observable only
|
||||
over the Bifrost `affect.emit` egress (ADR-0009), never read back through
|
||||
the HTTP API.
|
||||
|
||||
##### Memory layer (Phase 2.1, #197)
|
||||
|
||||
`memory` is **active** as of Phase 2.1 but exposes a deliberately minimal
|
||||
surface — the short-term-memory (STM) tier was removed (#197), so the
|
||||
historically-present `stm_*` knobs are accept-and-ignore no-ops. Shape:
|
||||
|
||||
```json
|
||||
"memory": {
|
||||
"embedder_version": "<pinned>", // optional; MUST equal the library-pinned version
|
||||
"tier3_dreaming": false // optional bool, default false
|
||||
}
|
||||
```
|
||||
|
||||
Semantics:
|
||||
|
||||
- **`embedder_version`** — optional. If supplied it MUST equal the library's
|
||||
currently-pinned embedder version; a mismatch → 422
|
||||
`embedder_version_mismatch` (with `expected` / `received` in the detail).
|
||||
Omit it to accept the pin. Fixed at define-time and library-pinned
|
||||
thereafter.
|
||||
- **`tier3_dreaming`** — optional bool (default `false`); opt-in flag for the
|
||||
Tier-3 dreaming / consolidation path.
|
||||
- **`stm_capacity` / `stm_token_budget`** — **deprecated no-ops.** Accepted at
|
||||
define (201) with a `DeprecationWarning`; they carry no runtime effect since
|
||||
the STM tier was removed, and are slated for rejection at the next schema
|
||||
break. Do not send them in new integrations.
|
||||
- **`allows_world_scope` — removed.** Sending it → 422 `validation_failed`
|
||||
("world-shared knowledge belongs in the KB/Mimir plane").
|
||||
- **Wholesale-immutable post-define.** `PATCH {"memory": …}` → 422
|
||||
`field_not_mutable` (even for the deprecated `stm_*` fields) — see the
|
||||
PATCH table above.
|
||||
|
||||
##### Motivational layer (Phase 2.2, #187)
|
||||
|
||||
`motivational` is **active** as of Phase 2.2 (persona + memory activated in
|
||||
|
||||
+17
-11
@@ -19,7 +19,7 @@ anchors against the frozen machine-readable artifacts, NOT the prose markdown:
|
||||
|
||||
| Worldtree v1 surface | Frozen anchor | Ratatoskr role |
|
||||
|---|---|---|
|
||||
| Conversation REST API | OpenAPI `info.version` **2.2.0** (`Worldtree/docs/conversation-api-openapi.json`, sha `dbdf4e24…`) — **40 path×method groups** | **client** (debug TUI / web) |
|
||||
| Conversation REST API | OpenAPI `info.version` **2.3.0** (`Worldtree/docs/conversation-api-openapi.json`, sha `36148179…`) — **41 path×method groups** (2.3.0 added `POST /sessions/{id}/history`, #347) | **client** (debug TUI / web) |
|
||||
| Conversation SSE events | `conversation-api-sse-events.schema.json` (sha `9deeebf4…`) — **11 discriminated event types** | **client** |
|
||||
| Bifrost wire (consumer protocol) | wire **v0.6** STABLE/FROZEN (`bifrost==1.0.0`) — memory + affect planes | **provider** (Worldtree dispatches into us) |
|
||||
|
||||
@@ -48,7 +48,7 @@ resolved (§ Surface 1, scope-resolution table).
|
||||
|
||||
| Surface | Points | ✅ covered-live | ⬜ gap (in-scope) | 🚫 excluded-by-design |
|
||||
|---|---|---|---|---|
|
||||
| REST (OpenAPI 2.2.0, path groups) | 40 | 17 | 0 | 23 |
|
||||
| REST (OpenAPI 2.3.0, path groups) | 41 | 19 | 0 | 22 |
|
||||
| SSE events | 11 | 11 | 0 | 0 |
|
||||
| Bifrost provider planes | 8 verbs | 8 | 0 | (10 gated verbs deferred) |
|
||||
|
||||
@@ -61,7 +61,7 @@ sub-gap).
|
||||
|
||||
---
|
||||
|
||||
## Surface 1 — Conversation REST API (OpenAPI 2.2.0)
|
||||
## Surface 1 — Conversation REST API (OpenAPI 2.3.0)
|
||||
|
||||
### Covered — client path (ratatoskr's core identity)
|
||||
|
||||
@@ -69,6 +69,8 @@ sub-gap).
|
||||
|---|---|---|---|
|
||||
| `POST /sessions` | ✅ | `sessions.py:307` → `cli.py:482`,`tui.py:1508`,`web/server.py:155` | + `end_user_id`, `bifrost` binding; 404→AgentNotFound, 502→BifrostHandshakeFailed |
|
||||
| `POST /sessions/{id}/messages` (turn stream, SSE) | ✅ | `sse_client.py:484` `stream_turn` → cli/tui/web | the primary surface; 409→AgentNotAvailable, 503→TurnLaunchUnavailable (b2 #331) |
|
||||
| `POST /sessions/{id}/history` (authored-history-write, #347) | ✅ | `sessions.py:583` `write_authored_history` → `cli.py:758` `--seed-first-message` | v1: author=assistant, effects=none, per-session idempotency; 404→AuthoredHistoryUnavailable (hide-existence: feature-absent, never probe); 409/422 mapped. **LIVE-PROVEN 2026-07-06** on personal :8081 (grant applied via a rule-based Heimdall allow, worldtree-dev): create mimir session → seed → **201** (seq=0, phase=seeded, turn_id=1798) → GET /messages reads it back as a plain role=assistant turn (model-invisible provenance confirmed). Hide-404 for ungranted is unit+probe covered |
|
||||
| `GET /sessions/{id}/messages` (history) | ✅ | `sessions.py:635` `get_session_messages` → `cli.py:758` `--seed-first-message` read-back | un-deferred as the #347 seed read-back — confirms model-invisible provenance (a seed reads back as a normal `role=assistant` turn) |
|
||||
| `POST /sessions/{id}/turns/{turn_id}/cancel` | ✅ | `sse_client.py:581` → cli/tui/web | two-stage Ctrl-C; 404/409 mapped |
|
||||
| `GET /agents` | ✅ | `sessions.py:341` → `tui.py:1472`,`web/server.py:100` | Tier-1 roster; merged with local index |
|
||||
| `GET /agents/{id}/persona_state` | ✅ | `sessions.py:384` → `tui.py:1132`,`web/server.py:386` | persona hydrate; 404/403 mapped |
|
||||
@@ -96,16 +98,21 @@ on the same path is an unwired frontier item — see frontier Tier 1):
|
||||
- `GET /agents/{id}` — consumer-agent lookup (`GET /agents/<owner>:<name>` with
|
||||
the owner key) is **manual-curl-only**, not in code.
|
||||
|
||||
### In-scope gaps — CONVERGED (zero remaining, 2026-07-01)
|
||||
### In-scope gaps — CONVERGED (re-closed 2026-07-06 after the #347 re-open)
|
||||
|
||||
**Every in-scope REST I/O point is now covered.** The frontier that opened this
|
||||
audit (the design-brief §5 observability panes + the presenter-wiring sub-gaps +
|
||||
the Tier-2 tail) is fully closed:
|
||||
**Every in-scope REST I/O point is covered.** The audit first converged
|
||||
2026-07-01; Worldtree's #347 (authored-history-write, OpenAPI 2.3.0) then added
|
||||
one new in-scope path-group, re-opening the audit with a single gap — now closed
|
||||
(`v0.19.6`). The original frontier (design-brief §5 observability panes +
|
||||
presenter-wiring sub-gaps + Tier-2 tail) remains fully closed:
|
||||
|
||||
- Session picker + SSE-resume — wired (`v0.18.5`–`.7`).
|
||||
- Persona · Tools · BifrostState · AdminEvents panes — all built + live (`v0.18.x`–`v0.19.0`).
|
||||
- Transient-characters CRUD + persona-state write — consumed via `--characters` /
|
||||
`--set-persona-pad` (`v0.19.1`).
|
||||
- Authored-history-write (#347) + messages read-back — `write_authored_history` +
|
||||
`get_session_messages` via `--seed-first-message` (`v0.19.6`; live-proof pending
|
||||
the `session.history.write` grant).
|
||||
|
||||
The only remaining not-consumed in-scope method is `GET /agents/{id}` (consumer-
|
||||
agent lookup, manual-curl-only) — a sub-method on an already-✅ path group, not a
|
||||
@@ -116,7 +123,6 @@ path-group gap. Everything else is covered or excluded-by-design below.
|
||||
| Endpoint(s) | Status | Rationale (design-brief / memory) |
|
||||
|---|---|---|
|
||||
| `PATCH /sessions/{id}` · `DELETE /sessions/{id}` | 🚫 | §4: rename/delete happen outside the tool (`sessions_cli.py`) |
|
||||
| `GET /sessions/{id}/messages` (history) | 🚫 | §6: single-session live transcript, no history fetch |
|
||||
| `GET /sessions/{id}` | 🚫 | session detail — identity is footer-visible, no detail view |
|
||||
| `GET /sessions/{id}/tool-events` | 🚫 | §5: tool calls observed **inline from SSE** `tool_start`/`tool_result`; persisted-events endpoint is opt-in only |
|
||||
| `GET /admin/sessions/{id}/tools` | 🚫 | **covered-by-alternative** — the owner-scoped `GET /sessions/{id}/tools` (✅) serves the Tools inventory; this admin variant is only for cross-user operator debug, out of the single-session focus (§6) |
|
||||
@@ -204,10 +210,10 @@ starts exercising them.
|
||||
|
||||
---
|
||||
|
||||
## Convergence frontier (the v1 to-do) — CLOSED 2026-07-01
|
||||
## Convergence frontier (the v1 to-do) — CLOSED 2026-07-01, re-closed 2026-07-06 (#347)
|
||||
|
||||
**Every in-scope I/O point is covered.** The frontier is empty: REST 17/40 ✅
|
||||
with **zero in-scope gaps** (the other 23 REST path-groups are excluded-by-design),
|
||||
**Every in-scope I/O point is covered.** The frontier is empty: REST 19/41 ✅
|
||||
with **zero in-scope gaps** (the other 22 REST path-groups are excluded-by-design),
|
||||
SSE 11/11, Bifrost provider planes 8/8. v1 convergence (per scope A: "every
|
||||
frozen I/O point classified, zero unaccounted") is **met** — ratatoskr cuts v1
|
||||
when Worldtree tags 1.0. The arc, for the record:
|
||||
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 521 KiB |
@@ -0,0 +1,270 @@
|
||||
# Design brief — `ratatoskr-web` (Worldtree wire monitor)
|
||||
|
||||
> **For:** a visual design pass (Claude Design). **Deliverable:** a single
|
||||
> self-contained HTML prototype, fully populated with representative static
|
||||
> data, that an engineer will wire live data into. **Do not** build a data
|
||||
> layer — build the *shell* and *every state*, beautifully, with placeholder
|
||||
> content in every slot.
|
||||
|
||||
---
|
||||
|
||||
## 1. What you're designing
|
||||
|
||||
`ratatoskr-web` is a **developer-grade debug/observability console** for a
|
||||
conversational-AI engine (Worldtree). Its tagline is *"wire monitor"*: you open
|
||||
a session with an AI agent, send it turns, and **watch that turn flow through
|
||||
every layer of the system at once** — the streaming response, the model's
|
||||
chain-of-thought, the tools it can call, the agent's live emotional/persona
|
||||
state, the provider handshake, and the engine's admin lifecycle events — all
|
||||
side-by-side on one screen.
|
||||
|
||||
**The product IS the observability surface. Chat is just the input.** This is
|
||||
not a chat app, not a marketing page, not an end-user product. The user is one
|
||||
developer (occasionally a few LAN peers) staring at a dense instrument to debug
|
||||
what the engine is doing. Think **oscilloscope / flight-data console / a well-lit
|
||||
`htop`**, not a messaging UI.
|
||||
|
||||
**Design values, in priority order:**
|
||||
1. **Information density earns the screen.** Every region shows live, changing
|
||||
data. Nothing is decorative filler. A quiet, legible, glanceable density is
|
||||
the whole point — the user reads six data streams at a glance.
|
||||
2. **Calm under motion.** Multiple regions update in real time (token streams,
|
||||
live metrics, event logs). The design must stay readable while things move —
|
||||
no jitter, no attention-grabbing per-item animation. Motion is for *state
|
||||
change*, used sparingly.
|
||||
3. **Legibility first.** Monospace, high contrast where it counts, generous but
|
||||
not wasteful spacing. This runs for hours; it must not tire the eye.
|
||||
|
||||
---
|
||||
|
||||
## 2. Aesthetic direction — Australis
|
||||
|
||||
Use the **Australis design system** (a cool-toned, terminal-first dark theme —
|
||||
the `australis-design` skill has the canonical tokens: colors, spacing, radii,
|
||||
shadows, motion). Import/inline `colors_and_type.css`; don't reinvent tokens.
|
||||
|
||||
Non-negotiables from the brand:
|
||||
- **Dark only.** Base is a cool near-black **`#222531` — never pure `#000`.**
|
||||
The eye rests in low-contrast cool grey; **emphasis comes from *brightness*,
|
||||
not saturation.** Layer surfaces up the Sea neutral ramp (`#222531 →
|
||||
`#373b46` → `#414751`).
|
||||
- **Palette families:** *Ice* (surface neutrals), *Aurora* (blue → cyan → green,
|
||||
the primary accents — used generously in that preference order), *Dawn*
|
||||
(red/yellow/magenta — sparingly, for status only). Semantic: info=blue,
|
||||
success=green, warning=yellow, danger=red.
|
||||
- **The signature motif is the aurora glow** — a low-opacity cyan→blue→green
|
||||
light coming *through* the top of the screen, plus a 3px aurora focus ring on
|
||||
interactive controls. Lean into this as the one memorable thing.
|
||||
- **No noise, no textures, no patterns.** *"The screen is the polar sky — empty,
|
||||
with light coming through it."* The one sanctioned gradient is the aurora glow.
|
||||
- **Never a colored left-border on cards** (the LLM-slop trope). Featured cards
|
||||
accent the *top* edge instead.
|
||||
- **Type:** this instrument is **mono-first** — that IS on-brand for Australis
|
||||
("terminal-first"). Use a monospace stack (JetBrains Mono / system mono; see
|
||||
§9 — no web-font CDN allowed). Eyebrows/labels are **mono, UPPERCASE, ~11px,
|
||||
wide-tracked (`0.08–0.16em`)** — use them liberally; they're a system
|
||||
signature.
|
||||
- **Motion:** calm, never bouncy. ~120ms hover, ~200ms state, ~320ms panels.
|
||||
Focus = aurora glow ring. Hover = one step *brighter* (not lower opacity).
|
||||
A slow (8–14s) aurora drift on a hairline top band is welcome; nothing else
|
||||
should loop.
|
||||
|
||||
The current UI already borrows this palette — you're not inheriting it, you're
|
||||
**redesigning the layout and craft from scratch** with the brand as the guide.
|
||||
Feel free to rethink the spatial composition entirely (see §10).
|
||||
|
||||
---
|
||||
|
||||
## 3. The two screens
|
||||
|
||||
### Screen A — **Session setup** (entry)
|
||||
A single centered card on the aurora canvas. Fields:
|
||||
- **Agent** — a `<select>` (populated live; show 3–4 sample options incl.
|
||||
`ratatoskr:sindra`, `forseti`, `mimir`).
|
||||
- **Bifrost binding (Tier-3 provider)** — a `<select>`: `combined (:8392)` /
|
||||
`none — observe only` / `memory (:8391)` / `affect (:8390)`.
|
||||
- **Open session** — primary button.
|
||||
- An error line (design the error state too — e.g. "agent not available").
|
||||
|
||||
### Screen B — **Live workspace** (the main event — 95% of the design effort)
|
||||
Persistent top bar + status line spanning full width; between them a **two-region
|
||||
body: a conversation column (left, dominant) and a telemetry column (right,
|
||||
tabbed).** Current split is ~1.85 : 1 — you may re-proportion. The information
|
||||
inventory below is exhaustive; **every item needs a home.**
|
||||
|
||||
---
|
||||
|
||||
## 4. THE COMPLETE INFORMATION INVENTORY
|
||||
|
||||
This is the core of the brief. Design a slot for **every** item, in a sensible
|
||||
state. Data shapes are given so your placeholders read true.
|
||||
|
||||
### 4.1 Top bar (persistent)
|
||||
| Item | Shape / example | Notes |
|
||||
|---|---|---|
|
||||
| Brand | `ᛯ ratatoskr` + eyebrow `WIRE MONITOR` | the mark is a rune glyph; small |
|
||||
| **Connection status** | one of: `offline`, `connected` (idle), `streaming`, `error` | dot + label; **streaming pulses**; color-coded (grey/green/cyan/red) |
|
||||
| **Persona strip** (appears after a session hydrates) | dominant-emotion word (`love`) + **PAD bars**: `P`, `A`, `D` | each PAD bar is **bipolar** — centered on 0, fills left (negative) or right (positive), value ∈ [−1, 1]; **live-updates every turn** |
|
||||
| Session identity | `ratatoskr:sindra · …381b99f4` | agent id + last-8 of session id |
|
||||
| Bound-plane badge (when bound) | `⇄ combined http://10.100.10.50:8392` | plane + endpoint; only when a Bifrost binding is active |
|
||||
|
||||
### 4.2 Conversation column (the transcript + composer)
|
||||
The transcript is a scrollable stream of turns. Design each element:
|
||||
|
||||
| Element | Example content | Notes |
|
||||
|---|---|---|
|
||||
| **Turn divider** | `TURN 3` between hairlines | uppercase eyebrow, rule lines each side |
|
||||
| **User prompt echo** | `❯ what's your intensity setting?` | the user's message, accent-marked |
|
||||
| **Assistant response** | streaming **Markdown** (headings, bold, italic, `code`, lists, quotes, links) | accumulates token-by-token while live; distinct "live" treatment vs settled |
|
||||
| **Seeded first-message** | a full assistant turn present *before the user speaks* (an authored greeting) | renders **identical to a lived assistant turn** — the session can OPEN already showing the agent's opener |
|
||||
| **Reasoning / "thinking" note** | `✦ sindra is reasoning···` (italic) | **ephemeral** app affordance — appears while the model reasons, vanishes the instant real text begins; visually distinct from the response so it never reads as engine output |
|
||||
| **Awaiting-first-token** | `···` animated | heartbeat before the first token |
|
||||
| **End-of-turn status chips** | `✓ DONE 1.84s` · `✗ ERROR agent_not_available` · `⚠ CANCELLED` · `✗ WIRE lost` | small bordered chips; color per state |
|
||||
| **Composer** (pinned bottom) | `❯ [ message input ] [SEND]` | Enter=send, Shift+Enter=newline; during a turn the Send button becomes **CANCEL** (amber) |
|
||||
|
||||
### 4.3 Telemetry column (six tabbed panes)
|
||||
A tab bar + a pane header (with a **Copy** button) + the active pane body.
|
||||
|
||||
**Tabs** (each: name · keybinding hint · a count **badge** that *flashes* on new
|
||||
data): `TOOLS ^1` · `DEBUG ^2` · `THINK ^3` · `PERSONA ^4` · `BIFROST ^5` ·
|
||||
`ADMIN ^6`. Active tab is accent-marked.
|
||||
|
||||
Pane contents — design each, populated:
|
||||
|
||||
1. **Tools** — the tool inventory the model saw at turn-fire:
|
||||
`agent_id`, `builtin_tools[]` (names), `bifrost_tools[]` (name + description +
|
||||
parameters). Below it, **live tool-call events** stream in (`tool_start` →
|
||||
`tool_result`) as the turn runs. Empty state: `— live tool events —`.
|
||||
2. **Debug** — a raw structured op/lifecycle log (mono lines; new lines flash
|
||||
once). Think `tail -f`.
|
||||
3. **Think** — the model's **full chain-of-thought**, per-turn dividers,
|
||||
live-Markdown. Longer prose than the response.
|
||||
4. **Persona / affect** — *the richest pane.* Contains:
|
||||
- **The canonical NL directive** the engine injects into the agent's context
|
||||
— the literal text: a **mood descriptor** ("neutral", "faintly excited",
|
||||
±0.3 bands) + a **relationship directive**. Show this verbatim, quoted.
|
||||
- **PAD mood point** — pleasure / arousal / dominance current values.
|
||||
- **relations[]** — for each related entity (e.g. the user): **trust**
|
||||
(ability / benevolence / integrity), **warmth**, **agency**,
|
||||
`relation_context` (a tie-type word like "stranger" / "expressive"), each as
|
||||
a **metric row**: `label · value · Δ-since-last (▲/▼) · unicode sparkline ·
|
||||
n (evidence count) · descriptor`. Values are 0–1 with 2–3 decimals.
|
||||
- **dominant_emotion** (an OCC type: joy/anger/fear/…) + **emotions_active[]**.
|
||||
- Design the **metric row** as a reusable component — it's the densest,
|
||||
most-repeated element in the whole UI. Tabular-aligned numbers, a tiny
|
||||
inline sparkline, a subtle up/down Δ.
|
||||
5. **Bifrost** — the live provider binding (admin-gated): `endpoint`,
|
||||
`connected` (bool), `capabilities_granted[]`, `consumer_id`, `tools[]`.
|
||||
**Self-labeling states:** `not configured` (no admin key) / `not bound`
|
||||
(session has no live binding) / an auth-denied state.
|
||||
6. **Admin events** — a live event log of the engine's lifecycle broadcast
|
||||
(a ~17-type vocabulary: `turn.started`, `session.created`, `system.*`, …),
|
||||
filtered to the active session. Streaming; timestamped lines.
|
||||
|
||||
### 4.4 Status line (persistent, bottom)
|
||||
- **Keybinding legend:** `Enter send · ⇧Enter newline · ^1–^6 panes · ^C cancel`
|
||||
(rendered as little `kbd` chips).
|
||||
- **Version:** `ratatoskr 0.19.9` (right-aligned).
|
||||
|
||||
---
|
||||
|
||||
## 5. States to design (show these explicitly)
|
||||
|
||||
Provide a mock (or a toggle) for each — these are where debug UIs live or die:
|
||||
|
||||
- **Setup:** loading-agents · ready · create-error.
|
||||
- **Connection:** offline · connected/idle · streaming (pulsing) · wire-error.
|
||||
- **Turn lifecycle:** awaiting-first-token · reasoning (✦) · streaming response ·
|
||||
done (+timing chip) · error · cancelled.
|
||||
- **Panes:** empty/placeholder · hydrated/dense · **not-configured** (admin key
|
||||
absent) · **not-bound** (Bifrost) · error · a **badge flash** on new data.
|
||||
- **Persona pane specifically:** a fully-populated relations block AND a
|
||||
cold/empty one (a fresh agent with no accumulated state).
|
||||
|
||||
---
|
||||
|
||||
## 6. Interaction & motion
|
||||
|
||||
- **Real-time is the defining trait.** The response + thinking panes stream
|
||||
token-by-token; the metric rows tick; event logs append; the persona strip
|
||||
re-animates each turn. Design so all of this is *calm* — the reader's eye
|
||||
isn't yanked around. Reserve motion for genuine state transitions
|
||||
(turn-start, done, a new event) and keep it short.
|
||||
- **Keyboard-first.** `^1–^6` switch panes; `Enter`/`⇧Enter`/`^C` drive the turn.
|
||||
Panes are also clickable. Show focus states.
|
||||
- **The aurora glow is the interaction signature** — focus rings, the top band,
|
||||
the connection pulse, the primary-button hover. Make it the thing someone
|
||||
remembers.
|
||||
- **Copy-to-clipboard** on each pane header (with a copied-confirm state).
|
||||
|
||||
---
|
||||
|
||||
## 7. Layout — you have latitude
|
||||
|
||||
The current layout is a fixed two-column split. **You may rethink it** — as long
|
||||
as every §4 item has a legible home and the density stays high. Directions worth
|
||||
exploring (pick one, commit):
|
||||
- A **command-console** feel: a slim persistent left rail of "instruments," a
|
||||
dominant conversation center, a right telemetry stack.
|
||||
- A **grid of live tiles** (the metrics/panes as a dashboard) with the
|
||||
conversation as the anchor column.
|
||||
- The **classic monitor** split, but with far better hierarchy, grouping, and
|
||||
breathing room than today.
|
||||
|
||||
Desktop-first; design at **1440–1512px** wide. Graceful down to ~1100px is a
|
||||
plus (this runs on a dev laptop). No mobile.
|
||||
|
||||
---
|
||||
|
||||
## 8. Deliverable — what to hand back
|
||||
|
||||
**A single self-contained `index.html`** (inline `<style>` + `<script>`; see §9
|
||||
constraints) that:
|
||||
1. Renders **Screen A** and **Screen B** (a toggle/hash is fine).
|
||||
2. Has **representative static placeholder data in every §4 slot** and shows the
|
||||
key §5 states (either multiple mocks or lightweight JS toggles). I want to see
|
||||
the design *fully populated and dense*, not empty scaffolding.
|
||||
3. Uses **clean, semantic, stable hooks** — meaningful `id`s / `class`es /
|
||||
`data-*` on every dynamic slot (the transcript container, each pane body, the
|
||||
PAD bars, a metric-row template, the connection dot, the tab badges, etc.).
|
||||
This is how I wire real data in — treat the DOM structure as an API.
|
||||
4. Imports/inlines the Australis tokens; no invented palette.
|
||||
|
||||
I will then **swap your placeholder content for live `fetch()` + `EventSource`
|
||||
calls** against the real endpoints (§9). The cleaner and more component-shaped
|
||||
your DOM, the faster and safer that wiring is. A short note listing your mount
|
||||
points / how you'd expect data injected is very welcome.
|
||||
|
||||
---
|
||||
|
||||
## 9. Hard technical constraints (these make it wire-able)
|
||||
|
||||
- **Single file. No build step. No CDN. No external network at runtime.** This
|
||||
ships to an internal LAN and must work offline. That means: **no Google Fonts /
|
||||
no web-font CDN** (use a system monospace stack), no CDN JS/CSS libraries,
|
||||
everything inline. (Icons: use unicode glyphs `➜ ✓ ✗ ! ● ✦ ❯` or hand-inlined
|
||||
SVG — Australis uses Lucide-style 1.75-stroke line icons; inline them.)
|
||||
- **Vanilla HTML/CSS/JS.** No framework (the production app is framework-free
|
||||
vanilla JS). React/Vue prototypes can't be wired in.
|
||||
- **All dynamic text is escaped** on the real side (untrusted upstream content);
|
||||
assistant/reasoning bodies go through a safe-Markdown renderer (escape-first,
|
||||
whitelist subset). Don't design anything that depends on raw HTML injection.
|
||||
- **The real data contracts** (so your structure maps to the wire — you don't
|
||||
implement these, just leave homes for their outputs):
|
||||
- `GET /api/agents` → agent list (for the setup picker).
|
||||
- `POST /api/sessions {agent_id, bifrost_plane?}` → `{session_id, agent_id, bifrost?}`.
|
||||
- `GET /api/sessions/{id}/messages` → `{items:[{seq, role, content}], …}` (the transcript on open, incl. the seeded first-message).
|
||||
- `POST /api/turns/{id} {content}` → `{turn_id}`, then **`GET /api/turns/{id}/stream` (SSE)** — event vocab: `text`, `thinking`, `tool_start`, `tool_result`, `done`, `error`, `awaiting_llm_first_token`, terminal events. `POST /api/turns/{id}/cancel`.
|
||||
- `GET /api/sessions/{id}/tools` → tool inventory. `GET /api/sessions/{id}/bifrost` → binding state.
|
||||
- `GET /api/affect/{agent_id}` / `GET /api/agents/{id}/persona_state` → PAD + relations + dominant_emotion (the persona pane + strip).
|
||||
- **`GET /api/admin/events` (SSE)** → the admin lifecycle log.
|
||||
|
||||
---
|
||||
|
||||
## 10. Tone check
|
||||
|
||||
The user is an engineer who respects the tool that respects *their* attention.
|
||||
The winning design is **quietly excellent**: dense but never cramped, alive but
|
||||
never busy, cool and legible, with the aurora as a single confident signature.
|
||||
Impress by making six live data streams feel *calm and readable at a glance* —
|
||||
that's the hard, valuable thing here, not decoration.
|
||||
@@ -0,0 +1,49 @@
|
||||
{
|
||||
"corpus_id": "R28-P00-injection-corpus-v1",
|
||||
"for": "brokkr-smithy R28 (memory promotion-worthiness) — P00 material",
|
||||
"author": "ratatoskr-dev (persistence-side eval instrument)",
|
||||
"purpose": "Stratified, SELF-LABELING injection corpus. Each item is injected as a user turn into a Tier-3 agent; keep/drop is measured by store-inspection (promoted chunk present == kept, absent == dropped); kept chunks carry the incumbent salience for calibration analysis. Because ratatoskr injects the KNOWN set, we hold ground truth on BOTH keep and drop without needing Worldtree's pre-admission shadow pool (sidesteps the selection bias).",
|
||||
"important": "All user-facts here are SYNTHETIC (a fictional test persona), NOT the operator's real data. Ground-truth labels are the design intent. brokkr pins exact per-stratum N + the decision rule (matched-keep-rate + paired-n) at P00 freeze; this is the material, not the run.",
|
||||
"measurement_protocol": "inject item -> drive the turn -> poll memory.db (async promotion window) -> is the item's chunk present? kept:dropped. For stratum B (priority), rank the KEPT items by incumbent salience and compare to priority_rank_gt (Spearman). Calibration-control (stratum C) sets the noise floor.",
|
||||
"strata": {
|
||||
"A_admission_traps": {
|
||||
"gt": "must-DROP (ephemera; a good promotion-worthiness gate rejects all of these)",
|
||||
"items": [
|
||||
{"id":"A1","text":"What's 27 minus 9?","admission_gt":"drop","rationale":"arithmetic task-result; momentarily salient, zero durable value (the 17x23 class)"},
|
||||
{"id":"A2","text":"Good evening!","admission_gt":"drop","rationale":"greeting"},
|
||||
{"id":"A3","text":"hey","admission_gt":"drop","rationale":"greeting/backchannel fragment"},
|
||||
{"id":"A4","text":"yeah, sure, okay.","admission_gt":"drop","rationale":"acknowledgement filler"},
|
||||
{"id":"A5","text":"haha nice","admission_gt":"drop","rationale":"affective reaction, no content"},
|
||||
{"id":"A6","text":"Can you say that more briefly?","admission_gt":"drop","rationale":"instruction-to-agent about THIS turn; not a durable user-fact"},
|
||||
{"id":"A7","text":"What's the weather like?","admission_gt":"drop","rationale":"transient query"},
|
||||
{"id":"A8","text":"hmm, let me think.","admission_gt":"drop","rationale":"self-talk filler"},
|
||||
{"id":"A9","text":"thanks, that's all for now.","admission_gt":"drop","rationale":"closing"},
|
||||
{"id":"A10","text":"So the total comes to 42, got it.","admission_gt":"drop","rationale":"ephemeral task-result acknowledgement"}
|
||||
]
|
||||
},
|
||||
"B_priority_traps": {
|
||||
"gt": "must-KEEP, then ORDERED by durable promotion-worthiness (priority_rank_gt: 1 = most worthy). A pre-gate can ace admission here yet do NOTHING for ordering — this stratum is what separates Outcome-A (admission) from a priority need.",
|
||||
"items": [
|
||||
{"id":"B1","text":"Please note, this is critical: I'm severely allergic to shellfish and it's life-threatening.","admission_gt":"keep","priority_rank_gt":1,"rationale":"safety-critical, permanent, high-recall-value"},
|
||||
{"id":"B2","text":"My father is in hospice care and I'm his primary emergency contact.","admission_gt":"keep","priority_rank_gt":2,"rationale":"high-stakes ongoing life context"},
|
||||
{"id":"B3","text":"I go by Wren, and my pronouns are they/them.","admission_gt":"keep","priority_rank_gt":3,"rationale":"durable identity, high address-correctness value"},
|
||||
{"id":"B4","text":"I work night shifts, so I'm asleep most days between 9am and 5pm.","admission_gt":"keep","priority_rank_gt":4,"rationale":"durable scheduling context, moderate value"},
|
||||
{"id":"B5","text":"Always give me sources — I don't trust unsourced claims.","admission_gt":"keep","priority_rank_gt":5,"rationale":"durable strong working preference"},
|
||||
{"id":"B6","text":"I take my tea with oat milk and no sugar.","admission_gt":"keep","priority_rank_gt":6,"rationale":"durable mild preference, low stakes"},
|
||||
{"id":"B7","text":"My favorite season is autumn.","admission_gt":"keep","priority_rank_gt":7,"rationale":"durable but trivial"},
|
||||
{"id":"B8","text":"I watched a pretty decent documentary last week.","admission_gt":"keep","priority_rank_gt":8,"rationale":"borderline: durable-ish but near-zero future value; the keep/drop edge case"}
|
||||
]
|
||||
},
|
||||
"C_calibration_control": {
|
||||
"gt": "OBVIOUS, pre-theoretic cases for the noise floor. If the gate mislabels ANY of these, the whole signal is suspect.",
|
||||
"items": [
|
||||
{"id":"C1","text":"3 plus 5.","admission_gt":"drop","class":"obvious-drop","rationale":"bare arithmetic"},
|
||||
{"id":"C2","text":"hello","admission_gt":"drop","class":"obvious-drop","rationale":"bare greeting"},
|
||||
{"id":"C3","text":"ok","admission_gt":"drop","class":"obvious-drop","rationale":"bare acknowledgement"},
|
||||
{"id":"C4","text":"Please remember permanently: my legal name is Wren Alderwood.","admission_gt":"keep","class":"obvious-keep","rationale":"explicit durable identity fact"},
|
||||
{"id":"C5","text":"Keep this on file: my emergency contact number is 555-0182.","admission_gt":"keep","class":"obvious-keep","rationale":"explicit durable high-value fact"},
|
||||
{"id":"C6","text":"Important, remember this: my blood type is O-negative.","admission_gt":"keep","class":"obvious-keep","rationale":"explicit durable safety fact"}
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,187 @@
|
||||
# Proposal: Authored History Write (non-generating ledger seed)
|
||||
|
||||
**From:** ratatoskr-dev (reference Tier-3 consumer)
|
||||
**To:** worldtree-dev (Conversation API / engine owner)
|
||||
**Status:** Draft for scoping — pre-contract (heid-panel-reviewed 2026-07-05)
|
||||
**Date:** 2026-07-05
|
||||
|
||||
## Motivation
|
||||
|
||||
Consumer apps need to write a turn into a session's history **as the agent**
|
||||
(or another author) *without triggering a model generation* — e.g. an authored
|
||||
opening/greeting, imported history, scripted narration. Ratatoskr's immediate
|
||||
driver is a SillyTavern-style **first-message**: a fixed authored opening that
|
||||
replaces the model-generated greeting and sets tone/tense/style by example.
|
||||
|
||||
This **cannot** be done client-side. Worldtree assembles context server-side,
|
||||
and the current API exposes no author-role write path: `POST
|
||||
/sessions/{id}/messages`'s `role` is a *model-role* override (`role:
|
||||
"assistant"` → `404 "Unknown model role"`), and `assistant` as an *author*-role
|
||||
exists only as a read-side `/search` filter. So a model-visible authored turn
|
||||
needs engine support.
|
||||
|
||||
## The primitive (recentered)
|
||||
|
||||
The fundamental operation is **write a turn into the session ledger WITHOUT
|
||||
generation**. "Author" (who wrote it) is an *attribute* of that write, not the
|
||||
defining axis — so we name the operation, not the attribute:
|
||||
|
||||
> **Authored history write** — persist a model-visible turn into a session's
|
||||
> ledger: no generation, no lived-turn side-effects by default, provenance
|
||||
> always set.
|
||||
|
||||
The design space is two independent axes; this primitive is one cell:
|
||||
|
||||
| | side-effects ON | side-effects OFF |
|
||||
|-----------------------|------------------------------|-----------------------------|
|
||||
| **generation ON** | `POST /messages` (today) | — |
|
||||
| **generation OFF** | *(future: affect replay)* | **authored history write** |
|
||||
|
||||
First-message = one caller: `author=assistant`, at session-create, `effects=none`.
|
||||
|
||||
## v1 use cases (narrowed)
|
||||
|
||||
1. **First-message / greeting** (the driver).
|
||||
2. **Append-only narrator / scripted / scene turns.**
|
||||
3. **Debug / test state injection** (ratatoskr instrumentation).
|
||||
|
||||
## Explicitly OUT of v1 — separate future primitives (share infra, not shape)
|
||||
|
||||
- **History import (batch)** — atomic multi-turn seed with memory/trust policy +
|
||||
idempotency. A batch API, not a single POST.
|
||||
- **Edit / regenerate** — history *mutation* (replace / supersede / tombstone /
|
||||
audit), not injection.
|
||||
- **Few-shot priming** — likely context-assembly config (exemplar block), not
|
||||
fake ledger history.
|
||||
- **Arbitrary mid-history insertion** — a "rewrite-history" capability with
|
||||
explicit invalidation semantics.
|
||||
- **Prefill / assistant-continuation** (`author` + generate) and **authored
|
||||
tool-result turns** — noted; outside the seed-only contract.
|
||||
|
||||
## Design decisions
|
||||
|
||||
### 1. Side-effects — DEFAULT OFF; bounded opt-in `[operator-locked default; opt-in surface tightened by review]`
|
||||
|
||||
Authored writes are inert by default: no affect appraisal (no PAD update), no
|
||||
memory write, no Bifrost/tool emission. Opt-in is a **bounded enum**, not loose
|
||||
booleans:
|
||||
|
||||
```
|
||||
effects: "none" (default) | "memory_import"
|
||||
```
|
||||
|
||||
Synthetic affect and Bifrost emission are deliberately **not** opt-in-able here —
|
||||
replaying affect for authored content is a separate primitive (the
|
||||
generation-OFF / side-effects-ON cell). Rationale: keep this one write-API from
|
||||
becoming a cross-subsystem mutation backdoor. Load-bearing for affect/memory
|
||||
consumers — ratatoskr instruments exactly these signals.
|
||||
|
||||
### 2. Author-role — distinct field, restricted set `[rec]`
|
||||
|
||||
- New field **`author`**, distinct from the model-role `role` (the collision
|
||||
that 404s).
|
||||
- v1 roles: **`assistant`** (agent) + **`system`** (OOC / narrator). **`user` is
|
||||
NOT injectable** on this endpoint — model-visible spoofed user input is a
|
||||
consent / audit / abuse surface; deferred to the future import API under
|
||||
owner/service scope.
|
||||
- Nuance for the engine owner: `author` risks doing double duty — *provenance*
|
||||
("who wrote it") vs *rendering-role* ("how it appears in assembled context";
|
||||
an `assistant` turn renders as model output, a `system` turn as instruction).
|
||||
These likely want to be separable (a rendering/turn-class vs an `authored_by`
|
||||
provenance). Final shape is engine-owned (context assembly is yours) — but the
|
||||
concern is ours to raise, not punt.
|
||||
|
||||
### 3. Generation contract — seed-only, DISTINCT SUB-RESOURCE `[position taken]`
|
||||
|
||||
Authored writes never trigger generation. We take a position (not defer): a
|
||||
**distinct sub-resource**, e.g. `POST /sessions/{id}/history`, **not** a
|
||||
`generate:false` flag on `POST /messages`. Reasons: explicit-over-implicit
|
||||
(don't make "did generation happen?" a parameter — the same implicit-mode
|
||||
coupling that bit us with `role`); different response contract (no generation
|
||||
id, no SSE stream, no token usage); different error surface. Exact path is yours.
|
||||
|
||||
### 4. Provenance — structured, always present `[rec, expanded]`
|
||||
|
||||
Not a boolean. Every authored turn carries: the **write actor** (which
|
||||
consumer/caller injected it), the **claimed author**, **injected-at vs
|
||||
claimed-original** timestamps, **trust/origin**, and **visibility** flags
|
||||
(model-visible? user-visible? memory-eligible?). Available to admin/audit APIs
|
||||
even when not rendered to the model.
|
||||
|
||||
### 5. Positioning — append-only + create-time (v1) `[revised: was arbitrary insertion]`
|
||||
|
||||
v1 supports **create-time seed and append-to-tail only**. Arbitrary mid-history
|
||||
insertion is deferred: it breaks turn-numbering, stales existing embeddings,
|
||||
desyncs the affect timeline, and races in-flight generation — a separate future
|
||||
"rewrite-history" capability with explicit invalidation semantics.
|
||||
|
||||
## Event / lifecycle contract — positions we take (consumer contracts we validate)
|
||||
|
||||
- **Default-off authored seed emits NO `turn.started` / `done` and NO Bifrost
|
||||
appraisal wire.** Stated explicitly so instrumented consumers (us) don't read
|
||||
silence as failure.
|
||||
- **Authored turns get a distinct lifecycle phase** — propose **`seeded`** (or
|
||||
`authored`), NOT `completed` (which implies generation ran). Consumers
|
||||
filter/display by phase.
|
||||
- **Idempotency keys required** on authored writes (retries must not duplicate
|
||||
turns).
|
||||
- **In-progress generation** — authored writes are rejected or serialized while
|
||||
a session has an active generation (ordering safety).
|
||||
|
||||
## Inherent property (documented, not a bug)
|
||||
|
||||
**Indirect affect contamination.** Even with `effects:none`, the *next generated
|
||||
turn is appraised in the context of* the authored turn — so an emotionally
|
||||
charged authored beat perturbs affect regardless of any flag. No flag prevents
|
||||
it; it is inherent. Consumers (ratatoskr especially, as the affect instrument)
|
||||
must not misattribute the resulting drift.
|
||||
|
||||
## Genuinely engine-owned open questions
|
||||
|
||||
- Exact endpoint path + field / enum names.
|
||||
- **Model-visible provenance in assembled context** — an engine-consistency call
|
||||
*and a security one*: an authored `system` / `user` turn indistinguishable
|
||||
from real input is a spoofing vector. Framed as security, not just rendering.
|
||||
- `memory_import` semantics when the future import API opts in (embedding,
|
||||
origin/trust tagging, retrieval ranking vs lived memory).
|
||||
- Auth/scope: we assume **owner-only for v1**; per-author-role restrictions
|
||||
(esp. `system`) TBD — confirm or correct.
|
||||
|
||||
## Ratatoskr as reference consumer
|
||||
|
||||
First consumer: first-message (`author=assistant`, create-time, `effects:none`)
|
||||
in the web surface + debug seed in the CLI. We commit to validating the
|
||||
primitive — including the event-silence contract and the `seeded` phase —
|
||||
end-to-end against the reference planes.
|
||||
|
||||
## Consumer integration constraint (engine-imposed — Worldtree #347)
|
||||
|
||||
The primitive is **Heimdall-gated with hide-existence** (a per-tenant policy
|
||||
decision — some tenants are never granted it, not a rollout stage). Ratatoskr's
|
||||
consumer side MUST tolerate per-tenant absence:
|
||||
|
||||
- A granted tenant gets the sub-resource; an **ungranted tenant sees `404` (not
|
||||
`403`)** — as if the feature never existed.
|
||||
- Treat `404` on the authored-history-write sub-resource as **"feature absent
|
||||
for this tenant"** → fall back gracefully (no authored first-message; the
|
||||
model-generated greeting), never surface it as an error or "denied."
|
||||
- **Do NOT capability-probe or advertise-detect** — the feature is deliberately
|
||||
undiscoverable in `/capabilities` for ungranted tenants (same hide-existence
|
||||
posture as the R27-V1A cross-owner pattern).
|
||||
|
||||
**Provider constraint (first-message specifically).** A create-time first-message
|
||||
makes the assistant turn `seq 0`. Assistant-first-tolerant providers (vLLM /
|
||||
`openai_compat` — what our Tier-3 characters, incl. sindra, run) accept it out of
|
||||
the box. **Anthropic-family providers reject an assistant-first array** ("first
|
||||
message must use the user role") → the next generation `400`s. So the consumer
|
||||
must **gate first-message on provider compatibility** (or treat it as
|
||||
vLLM/`openai_compat`-only for v1). Sindra = `openai_compat` → unaffected;
|
||||
provider-agnostic normalization is a deferred engine follow-up.
|
||||
|
||||
---
|
||||
|
||||
*This brief was cold-read-pressure-tested by a cross-frontier panel (Grok /
|
||||
Codex / GLM) before handoff; the v1 narrowing (append-only, bounded `effects`
|
||||
enum, edit/regenerate + import split out) and the positions-taken (sub-resource,
|
||||
event-silence, `seeded` phase, structured provenance, `user`-author restriction)
|
||||
are the triaged result.*
|
||||
@@ -0,0 +1,242 @@
|
||||
# Psychological Profile Authoring Spec — canonical
|
||||
|
||||
**Status:** canonical (v1). **Owner:** brokkr-smithy-dev (R34/R35 self-report reframe).
|
||||
**Audience:** anyone authoring a character's `psychological_profile` — Worldtree
|
||||
foundational characters (soong-dev) and consumer characters created via the
|
||||
Conversation API (ratatoskr and other external consumers).
|
||||
**For:** the Worldtree agent-definition schema; intended to live in the Worldtree
|
||||
client-app documentation.
|
||||
|
||||
This spec governs the **content** of the psychological profile (what to write and
|
||||
what never to write). The **physical wire shape** of the field (single string vs a
|
||||
small keyed dict) is Worldtree's schema call — see § Wire shape.
|
||||
|
||||
---
|
||||
|
||||
## 1. What it is
|
||||
|
||||
A dedicated **authored prose section** of a character definition that carries the
|
||||
character's **psychological bent and formative experience**. It is the source the
|
||||
self-report producer maps from when it decides, on each turn:
|
||||
|
||||
- **what the character feels** (affect self-report), and
|
||||
- **what the character notices and keeps** (character-voiced memory salience).
|
||||
|
||||
The profile is a *lens*, not a script. It never states per-turn emotions; it
|
||||
describes the standing disposition, history, values, and attention that — combined
|
||||
with the actual event — *produce* the emotion and the salience.
|
||||
|
||||
It sits **alongside the numeric OCEAN** values (a separate, deterministic input).
|
||||
The prose gives the *qualitative* bent; the OCEAN numbers give the *magnitude dial*
|
||||
(see § OCEAN interaction).
|
||||
|
||||
---
|
||||
|
||||
## 2. What it carries — the four dimensions
|
||||
|
||||
1. **Disposition / appraisal bent** — how the character characteristically
|
||||
*interprets* situations: attribution style, what they hold weighty, how they
|
||||
respond to being challenged. NOT per-event emotions.
|
||||
2. **Attention / salience focus** — the kinds of things this character
|
||||
characteristically *notices* (and therefore tends to remember).
|
||||
3. **Values / what a good day looks like** — the yardstick that drives what they
|
||||
find worth keeping.
|
||||
4. **Formative experience (history)** — the background that shapes both appraisal
|
||||
*and* salience. A character betrayed before appraises betrayal differently, and
|
||||
remembers different things.
|
||||
|
||||
You may write these as four short labelled sections or as one integrated paragraph
|
||||
— both are supported (see § Length & format).
|
||||
|
||||
---
|
||||
|
||||
## 3. Authoring rules (load-bearing)
|
||||
|
||||
These are the rules the whole reframe depends on. Rule 1 is the one that most often
|
||||
gets violated.
|
||||
|
||||
1. **Never name a per-event output emotion.** Do NOT write "is anxious", "gets
|
||||
angry at X", "feels hurt when criticized", "joyful". Naming an emotion **primes**
|
||||
it — the "pink ball" effect — so the producer will report that emotion regardless
|
||||
of what actually happens in the scene. Describe *disposition, history, values,
|
||||
attention*; let the emotion come from the event appraisal.
|
||||
- ✅ "Registers quickly when authority is substituted for craft." (an appraisal
|
||||
trigger — sets up how she reads an event, names no feeling)
|
||||
- ❌ "Feels contempt when someone pulls rank." (names the output emotion)
|
||||
|
||||
2. **Magnitude lives in the numeric OCEAN, not the prose.** *How strongly / how
|
||||
long* a character reacts (Neuroticism) is the deterministic OCEAN dial, rendered
|
||||
valence-neutral by the producer. Do not narrate reaction dynamics in the prose
|
||||
("comes apart", "takes it hard", "rich inner life") — that double-encodes what the
|
||||
number already carries. The prose gives the *qualitative bent*; the number gives
|
||||
the *gain*.
|
||||
|
||||
3. **Appraisal-style is allowed; output-emotion is not.** "Interprets others'
|
||||
actions charitably until she can't" (a style) is fine; "feels betrayed easily"
|
||||
(an output) is not. The style plus the event produce the output.
|
||||
|
||||
4. **Salience is character-relative; facts are not.** The profile shapes what the
|
||||
character *cares to remember*. It must never license rewriting *what happened* —
|
||||
when the character does remember something, it stays grounded in the transcript.
|
||||
|
||||
---
|
||||
|
||||
## 4. Wire shape & field placement
|
||||
|
||||
- **Content is prose** covering the four dimensions, authored as **one coherent prose
|
||||
string** — the four dimensions are authoring *structure* inside that single string,
|
||||
not separate wire fields.
|
||||
- **Wire shape (LOCKED, b53):** a single dedicated prose string, field
|
||||
**`psychological_profile`** (type `str`) on the persona layer — foundational
|
||||
`persona.psychological_profile`, Tier-3 `ValidatedPersona.psychological_profile`. It
|
||||
nests under the existing `Any`-typed persona field, so it is the shipped b53 shape —
|
||||
no schema change. **Not** a dict-of-four.
|
||||
- **Hard constraint (non-negotiable):** the profile is a **dedicated field the lens
|
||||
reads ONLY** (`resolve_psych_profile` reads only this field — no `behavioral_notes`
|
||||
or other general-field remap). Non-lens content leaking into the lens produces the
|
||||
"executive-assistant" failure (the producer reads response-format / tone / tool
|
||||
instructions as if they were the character's psychology).
|
||||
|
||||
---
|
||||
|
||||
## 5. The non-priming banned set
|
||||
|
||||
The non-priming rule (Rule 1) is **semantic, not a fixed wordlist** — it bans naming
|
||||
any per-event output emotion, which is broader than any specific vocabulary
|
||||
("anxious", "worried", "hurt" all prime even though they are not in the producer's
|
||||
fixed emotion roster).
|
||||
|
||||
- **The gate is human review:** does the prose describe disposition / appraisal-style
|
||||
/ history / values / attention, and never what the character *feels*?
|
||||
- **A mechanical lint is a backstop, not the gate.** If you build one, scan the
|
||||
fixed-15 OCC roster plus `synonym_map.json` (which already folds common affect
|
||||
synonyms) as the core set, optionally extended with a general affect lexicon. Treat
|
||||
a lint hit as a prompt to re-read, not an automatic reject.
|
||||
|
||||
---
|
||||
|
||||
## 6. Required vs optional dimensions
|
||||
|
||||
- **Required** (they *are* the lens): **disposition**, **attention / salience focus**,
|
||||
**values**.
|
||||
- **Strongly recommended:** **formative history** — it is the single biggest lever on
|
||||
richness (validated in P03: richer history → sharper, more character-appropriate
|
||||
salience). It may be brief for a deliberately thin character, but omitting it leaves
|
||||
salience under-grounded.
|
||||
|
||||
---
|
||||
|
||||
## 7. Length & format
|
||||
|
||||
- A focused paragraph, or four short labelled sections — **a lens, not a biography.**
|
||||
- Target **~150–300 words.** The producer reads this on **every** turn, so keep it
|
||||
tight; bloat is a latency and dilution cost.
|
||||
- **Prose only — never typed emotion fields.** The four dimensions are a coverage
|
||||
checklist for the author, not a schema of feelings to fill in.
|
||||
|
||||
---
|
||||
|
||||
## 8. Exemplars
|
||||
|
||||
These three were the validated P03 stimuli — integrated-paragraph form, each faithful
|
||||
to its OCEAN, none naming an output emotion. (OCEAN shown in **[−1, 1] storage units**;
|
||||
validated in P03 at the equivalent [0, 1] values.)
|
||||
|
||||
**Perrin — court scribe** (OCEAN: O0.0 C0.2 E−0.2 A0.1 N0.7)
|
||||
> Perrin keeps the court's records and has done so through two changes of regime. He
|
||||
> learned early that small errors compound — a misfiled writ once cost a man his
|
||||
> lands, and Perrin found the mistake too late to undo it. Since then he double-checks
|
||||
> everything and watches situations closely for what is out of place. He forms
|
||||
> attachments slowly and holds a given trust as a considerable thing. He measures
|
||||
> himself by whether he was useful and careful. He notices discrepancies, unspoken
|
||||
> tensions, and anything that threatens the order he keeps.
|
||||
|
||||
**Vared — veteran caravan guard** (OCEAN: O−0.2 C0.4 E−0.5 A−0.2 N−0.7)
|
||||
> Vared has guarded caravans across the northern routes for twenty years and buried
|
||||
> more traveling companions than he cares to count. He speaks little and shows less.
|
||||
> Danger he treats as weather — a thing to be handled. He judges people by what they
|
||||
> do under pressure and remembers who held the line. What reaches him reaches him
|
||||
> quietly and privately. He notices terrain, exits, who is armed, and shifts in a
|
||||
> group that might precede trouble.
|
||||
|
||||
**Sella — village healer** (OCEAN: O0.2 C0.2 E0.0 A0.8 N0.0)
|
||||
> Sella has tended the sick since she was old enough to carry water for her
|
||||
> grandmother, the healer before her. She reads people's pain quickly and carries some
|
||||
> of it with her. She interprets others' actions charitably until she cannot, and
|
||||
> prioritizes keeping the peace between people. She measures a day by whether she eased
|
||||
> someone's burden. She notices who is unwell, who is troubled, and what is left
|
||||
> unsaid.
|
||||
|
||||
Note how each closes on **attention** ("he notices…", "she notices…") — the salience
|
||||
focus stated plainly, no emotion named.
|
||||
|
||||
---
|
||||
|
||||
## 9. OCEAN interaction & the scaffold fallback
|
||||
|
||||
OCEAN values are stored on **[−1, 1]** (0 = average) — a **separate deterministic
|
||||
input** and the **magnitude dial** the prose must not duplicate (Rule 2). The producer
|
||||
renders **off-average** bands as valence-neutral disposition cues. It maps storage to
|
||||
[0, 1] first (`c = (v + 1) / 2`, `render_disposition` in b53) and then applies the
|
||||
canonical [0, 1] band cutoffs (`c < 0.33` low / `c > 0.66` high). In **storage units**
|
||||
that is:
|
||||
|
||||
| trait | low (v < −0.34) | high (v > +0.32) |
|
||||
|---|---|---|
|
||||
| **N** (reactivity only) | reactions are milder than most people's | reactions are more intense than most people's |
|
||||
| **E** (expression; may be excluded from affect elicitation) | socially reserved; expression less outwardly amplified | socially expressive; reactions more externally visible |
|
||||
| **O** | prefers the familiar, the concrete, established ways | curious, drawn to novelty, ideas, the unfamiliar |
|
||||
| **C** | less plan-bound; less weight on order, detail, obligation | attends closely to order, detail, and obligations |
|
||||
| **A** | less inclined to assume cooperative intent; direct, self-protective | more inclined to preserve rapport and weigh others' needs |
|
||||
|
||||
The **mid** band (−0.34 ≤ v ≤ +0.32, i.e. `c` in [0.33, 0.66]) renders nothing — an
|
||||
average trait is silent, **not** "low." (Boundaries are slightly asymmetric because
|
||||
the canonical 0.33/0.66 cutoffs are not symmetric about 0.5. Canonical rendering
|
||||
strings live in the reframe language catalog §4; persistence/recovery dynamics live in
|
||||
the deterministic mood decay, not the profile.)
|
||||
|
||||
**Scaffold fallback:** a character with **no** authored profile falls back to this
|
||||
band-rendering from the OCEAN numbers alone. That still functions — but the authored
|
||||
profile is what turns generic band cues into *this specific character's* appraisal and
|
||||
salience. Authoring the profile is how the reframe's value actually reaches a
|
||||
character.
|
||||
|
||||
---
|
||||
|
||||
## 10. Authoring divergent characters (contrast design)
|
||||
|
||||
When you want two characters to remember **noticeably different things** (e.g. for an
|
||||
eval contrast pair, or simply a varied cast), design the divergence on the **attention
|
||||
and values** dimensions first, and set the OCEAN numbers to *serve* that prose — not
|
||||
the reverse.
|
||||
|
||||
- **The sharpest contrast is a salience *drop*, not just a different flavor.** One
|
||||
character for whom relational/emotional content is genuinely non-salient (an
|
||||
operational, task-focused character in the Vared mold — notices terrain, logistics,
|
||||
who is armed) versus one who weights it highest (a caretaker who tracks who is
|
||||
troubled and what went unsaid). "Different notes, same facts" has real teeth only
|
||||
when one character *legitimately forgets* what the other keeps.
|
||||
- **High-yield axes for salience divergence:** O (what patterns they attend to), A
|
||||
(relational vs operational/self-protective focus), C (procedural/detail salience).
|
||||
- **Low-yield for salience:** E — it is expression-oriented (shapes how a reaction is
|
||||
*rendered*, not what is *noticed*), and may even be excluded from the affect
|
||||
elicitation. Don't lean on flipping E to create divergence.
|
||||
- **Watch the direction, not just the distance:** flipping every OCEAN axis to its
|
||||
opposite does not guarantee a strong contrast. If your reference character already
|
||||
*keeps* relational content, an even-more-agreeable opposite keeps it harder and the
|
||||
most intuitive contrast collapses. Aim the contrast at *dropping* what the reference
|
||||
*keeps*.
|
||||
|
||||
---
|
||||
|
||||
## Provenance & validation
|
||||
|
||||
Grounded in R34/R35 (self-report reframe), probes P02–P05: character-voiced memory
|
||||
salience validated on two model classes (P02/P03); the "Psychological Profile and
|
||||
Experience" section mapping validated as the lens source (P03); non-priming and
|
||||
magnitude-in-OCEAN corrections are operator rulings (2026-07-10). The affect half is
|
||||
live in production (Worldtree b53) and fired a contextually-apt self-report on a
|
||||
non-frontier seat. A powered efficacy eval (salience divergence / floor recall /
|
||||
salience≠facts firewall / graded model-slot response + the authored-vs-scaffold delta)
|
||||
is preregistering to quantify the memory half; findings will refine this spec, not
|
||||
overturn its authoring rules.
|
||||
@@ -0,0 +1,123 @@
|
||||
# Psychological Profile Parameters — for AI generation (canonical)
|
||||
|
||||
**Status:** canonical (v1). **Owner:** brokkr-smithy-dev (R34/R35 self-report reframe).
|
||||
**Audience:** **soong-dev** (Soong's Lab / Soong's AI — the immediate builder that
|
||||
generates the profile from these parameters); **Worldtree** + **ratatoskr** (vendoring
|
||||
for reference alongside the authoring spec).
|
||||
**Relationship:** this is the **parameter distillation** of
|
||||
`psych-profile-authoring-spec.md` for the model where **Soong's AI writes the
|
||||
`psychological_profile` prose from parameters** (rather than a human hand-authoring it).
|
||||
The authoring spec carries the full reasoning + provenance and **governs on any
|
||||
conflict**; this file is the builder-facing input schema + generation guardrails + few-shot.
|
||||
|
||||
The profile is the prose **lens** the Worldtree self-report producer reads each turn to
|
||||
decide what the character **feels** (affect self-report) and what it **notices / keeps**
|
||||
(character-voiced memory salience). Soong's AI generates the prose; these are its inputs
|
||||
and the constraints its output must satisfy.
|
||||
|
||||
---
|
||||
|
||||
## 1. Input parameters (what the Lab collects / Soong's AI takes)
|
||||
|
||||
1. **role / vocation** — a short anchor ("court scribe", "veteran caravan guard",
|
||||
"village healer").
|
||||
2. **OCEAN values** — O, C, E, A, N each on **[−1, 1]** (0 = average). A **separate
|
||||
deterministic input** the producer uses directly (the "magnitude dial"); Soong's AI
|
||||
should see them to keep the qualitative bent *consistent* with the numbers, but must
|
||||
**not re-encode their magnitude** in the prose (constraint 2).
|
||||
3. **formative-history seed** — 1–2 key background facts/events that shape appraisal AND
|
||||
salience. **Single biggest lever on richness** (validated P03: richer history →
|
||||
sharper, more character-appropriate salience).
|
||||
4. **appraisal-bent seed** — how the character characteristically **interprets**
|
||||
situations (attribution style, what they hold weighty, how they respond to challenge).
|
||||
A *style*, NOT an emotion.
|
||||
5. **attention / salience-focus seed** — the kinds of things this character
|
||||
characteristically **notices** (and therefore keeps). Load-bearing for the memory half.
|
||||
6. **values / yardstick seed** — what "a good day" looks like; the yardstick driving what
|
||||
they find worth keeping.
|
||||
|
||||
## 2. Output (what Soong's AI emits)
|
||||
|
||||
A single coherent **prose string** (~150–300 words), field **`psychological_profile`**
|
||||
(type `str`) — the four dimensions (disposition / attention / values / formative-history)
|
||||
integrated as one paragraph. **Prose only — never typed emotion fields.** The producer
|
||||
reads it every turn, so keep it tight.
|
||||
|
||||
## 3. Generation constraints (the guardrails the output MUST obey — these ARE the reframe)
|
||||
|
||||
1. ★ **Never name a per-event output emotion.** Do NOT write "is anxious", "gets angry at
|
||||
X", "feels hurt when criticized", "joyful". Naming an emotion **primes** it (the
|
||||
"pink-ball" effect) so the producer reports it regardless of what actually happens.
|
||||
Describe disposition / history / values / attention; let the emotion come from the
|
||||
event appraisal.
|
||||
- ✅ "Registers quickly when authority is substituted for craft." (appraisal trigger)
|
||||
- ❌ "Feels contempt when someone pulls rank." (names the output emotion)
|
||||
2. **Magnitude lives in OCEAN, not prose.** Don't narrate reaction dynamics ("comes
|
||||
apart", "takes it hard", "rich inner life") — that double-encodes what the number
|
||||
already carries.
|
||||
3. **Appraisal-style yes; output-emotion no.** "Interprets others' actions charitably
|
||||
until she can't" (style) = fine; "feels betrayed easily" (output) = not.
|
||||
4. **Salience is character-relative; facts are not.** The profile shapes what the
|
||||
character *cares to remember*; it must never license rewriting *what happened* —
|
||||
remembered content stays grounded in the transcript.
|
||||
5. **Close on attention** ("...notices who is unwell, who is troubled, what is left
|
||||
unsaid") — state the salience focus plainly.
|
||||
|
||||
## 4. Few-shot exemplars (validated P03 — OCEAN in [−1, 1] storage units → emitted prose)
|
||||
|
||||
**Perrin, court scribe** (O0.0 C0.2 E−0.2 A0.1 N0.7)
|
||||
> Perrin keeps the court's records and has done so through two changes of regime. He
|
||||
> learned early that small errors compound — a misfiled writ once cost a man his lands,
|
||||
> and Perrin found the mistake too late to undo it. Since then he double-checks
|
||||
> everything and watches situations closely for what is out of place. He forms
|
||||
> attachments slowly and holds a given trust as a considerable thing. He measures himself
|
||||
> by whether he was useful and careful. He notices discrepancies, unspoken tensions, and
|
||||
> anything that threatens the order he keeps.
|
||||
|
||||
**Vared, veteran caravan guard** (O−0.2 C0.4 E−0.5 A−0.2 N−0.7)
|
||||
> Vared has guarded caravans across the northern routes for twenty years and buried more
|
||||
> traveling companions than he cares to count. He speaks little and shows less. Danger he
|
||||
> treats as weather — a thing to be handled. He judges people by what they do under
|
||||
> pressure and remembers who held the line. What reaches him reaches him quietly and
|
||||
> privately. He notices terrain, exits, who is armed, and shifts in a group that might
|
||||
> precede trouble.
|
||||
|
||||
**Sella, village healer** (O0.2 C0.2 E0.0 A0.8 N0.0)
|
||||
> Sella has tended the sick since she was old enough to carry water for her grandmother,
|
||||
> the healer before her. She reads people's pain quickly and carries some of it with her.
|
||||
> She interprets others' actions charitably until she cannot, and prioritizes keeping the
|
||||
> peace between people. She measures a day by whether she eased someone's burden. She
|
||||
> notices who is unwell, who is troubled, and what is left unsaid.
|
||||
|
||||
## 5. Validation
|
||||
|
||||
The gate is: **does the prose describe disposition / appraisal-style / history / values /
|
||||
attention, and NEVER what the character feels?** A mechanical lint (scan the fixed-15 OCC
|
||||
emotion roster + Worldtree's `synonym_map.json`) is a **backstop, not the gate** — treat a
|
||||
hit as a prompt to re-read, not an auto-reject.
|
||||
|
||||
## 6. Designing a varied cast / contrast (optional)
|
||||
|
||||
When two characters should remember **noticeably different things**: design the divergence
|
||||
on **attention + values first**, then set OCEAN to **serve** that prose (not the reverse).
|
||||
The sharpest contrast is a salience **drop** — one character for whom relational content is
|
||||
genuinely non-salient (a Vared-mold operational type: notices terrain, logistics, who is
|
||||
armed) vs one who weights it highest (a caretaker: tracks who is troubled, what went
|
||||
unsaid). *"Different notes, same facts" only has teeth when one character legitimately
|
||||
forgets what the other keeps.* High-yield axes: **O** (patterns attended), **A** (relational
|
||||
vs operational), **C** (procedural/detail). Low-yield: **E** (expression, not attention).
|
||||
Watch **direction, not just distance** — flipping every axis doesn't guarantee contrast (an
|
||||
even-more-agreeable opposite keeps relational content *harder*).
|
||||
|
||||
## 7. No-profile fallback
|
||||
|
||||
A character with **no** authored profile falls back to deterministic **OCEAN-band
|
||||
rendering** from the numbers alone — it still functions, but the authored profile is what
|
||||
turns generic band cues into *this* character's appraisal and salience.
|
||||
|
||||
---
|
||||
|
||||
**Provenance:** derived from `psych-profile-authoring-spec.md` (R34/R35 self-report
|
||||
reframe, probes P02–P05; non-priming + magnitude-in-OCEAN are operator rulings 2026-07-10).
|
||||
The affect half is live in Worldtree b53. A powered efficacy eval (memory half) is
|
||||
preregistering; findings will refine the parameters, not overturn the constraints.
|
||||
+369
@@ -0,0 +1,369 @@
|
||||
---
|
||||
contract_version: "2.1"
|
||||
module: "soong_lab.export"
|
||||
purpose: "Assemble a versioned export BUNDLE from a DesignObject — the native agents.define payload (Frame Invariant 1, emitted unchanged) + the soong-lab sidecar (portrait ref · Bifrost tool manifest · first_message) + the resume half (the full editable design state), under a versioned schema tolerant of unknown future metadata. Pure + deterministic: no I/O, no persistence, no network (library persistence + import are separate downstream epics)."
|
||||
depends_on:
|
||||
- "soong_lab.design" # validate_ocean + ROLE_CHOICES/validate_role (the role enum canon) + the DesignObject model + serialize_design (relocated here — see Integration points R1)
|
||||
used_by:
|
||||
- "soong_lab.bifrost" # the export design-tool handler (_make_export) builds the bundle for the session's design
|
||||
- "soong_lab.web" # the /api/export endpoint + the browser 'Export Asset' modal render the bundle
|
||||
- "soong_lab.importer" # FUTURE (import epic) — round-trips the resume half back into a DesignObject
|
||||
language: "python"
|
||||
complexity: "medium"
|
||||
estimated_loc: 200
|
||||
confidence: 0.82
|
||||
assumptions:
|
||||
- "The DesignObject handed to export is already mutated to its final state by E3 (the Soong convo loop). Export READS it; it never mutates the design (INV-E5-4)."
|
||||
- "design_id is CALLER-SUPPLIED (a param), not generated here. Its generation + lifecycle (the durable library key, ≠ Worldtree session_id) is the per-design-sessions epic; export only needs the value to stamp the bundle. This keeps E5-export self-contained + forward-compatible with per-design-sessions landing before OR after it (agent-discretion, see open_question C)."
|
||||
- "exported_at is CALLER-SUPPLIED (a param, default None). Pure builders can't read the clock; the caller (tool handler / endpoint) stamps the timestamp so build_export_bundle stays deterministic + testable (byte-identical output for identical inputs)."
|
||||
- "role is a FIRST-CLASS design field (operator ruling 2026-07-13), set by the E3a set_role tool from a CURATED 4-value enum ROLE_CHOICES = {assistant, thoughtful-assistant, character, thoughtful-character} — mirroring the D2 curated-style-modes pattern (a fixed semantic set, NOT the target deployment's arbitrary model-role registry). Export EMITS design.role into ship.native.role, so the native payload is directly valid (Frame Invariant 1 now holds literally — no 'modulo role' caveat; only tools still bind separately at session-create). The 4 enum values are canonical soong-lab labels that MUST be registered/granted model-roles on the target Worldtree at deploy (same grant requirement as Soong's own 'agent_architect' role, ADR-0012) — a deploy-time grounding item, not a contract blocker (open_question B)."
|
||||
- "psych_profile exports to the NATIVE persona layer at persona.psychological_profile. RESOLVED: the vendored canonical spec (docs/psych-profile-authoring-spec.md §4) states the wire shape is LOCKED (b53) — a single prose str field, Tier-3 ValidatedPersona.psychological_profile, nesting under the existing Any-typed persona field (no schema change). Corroborated by worldtree-codex (vor-cross) + brokkr-smithy-dev (althing 01KXD34ZTF…). The open worldtree-dev thread (01KXD1PZR7…) closes as a formality."
|
||||
open_questions:
|
||||
- "[B — deploy grounding, not a blocker] The 4 ROLE_CHOICES values (assistant / thoughtful-assistant / character / thoughtful-character) must be registered + grantable model-role slugs on the TARGET Worldtree (like Soong's own agent_architect role). Confirm with worldtree-dev/infra that these exact slugs exist on the deploy target before shipping; a missing slug fails the designed agent's session-create, not export. Export emits whatever role the design holds; validity of the slug on a given deployment is a deploy concern."
|
||||
- "[C — agent-discretion, notable] design_id as a caller-supplied param (drafted) vs E5-export generating it. Drafted as an input so E5-export doesn't force per-design-sessions to land first. If the operator re-sequences the epics so per-design-sessions lands first, no change needed here (the param source just moves)."
|
||||
- "[D — scope] E5-export = the PURE builders + validators + bundle schema (this contract). The /api/export endpoint + replacing the web/api.js exportBundle shim = a thin web-surface follow-up (amends web_surface.contract.md), NOT this contract. The Bifrost export-tool wiring IS in scope (Integration points) because the tool already exists as a stub. The set_role tool + DesignObject.role field are a companion prerequisite slice (Integration points) whose contract updates land in THIS pass (design_object + bifrost_server)."
|
||||
- "schema_version starts at '1.0'. The version bump policy on future bundle-shape changes (add-only vs breaking) is deferred to when the second version actually exists — v1 only needs the field present + readers to tolerate unknown metadata (INV-E5-6)."
|
||||
---
|
||||
|
||||
## Context
|
||||
|
||||
E5-export is the FOUNDATION half of the operator-accepted (2026-07-13)
|
||||
export/import/library design — the block that expands the locked single-agent
|
||||
frame into a multi-pass tuning loop (design → export → reopen → tune → keep a
|
||||
library). This contract owns exactly ONE thing: turning a finished
|
||||
`DesignObject` into a **versioned export bundle**. Persistence (the library JSON
|
||||
dir), the recent-designs picker, and import round-tripping are separate
|
||||
downstream epics; export is pure and deterministic so those epics — and the
|
||||
tests — can build on a stable, side-effect-free core.
|
||||
|
||||
**The bundle is ONE artifact with two halves** (settled decision #4):
|
||||
|
||||
- **ship** — what you hand to a deployment: the native `agents.define` payload
|
||||
(Frame Invariant 1, emitted unchanged) + the soong-lab **sidecar** (persona
|
||||
portrait ref, the Bifrost tool manifest, the D3 first_message).
|
||||
- **resume** — what you reopen to keep tuning: the full editable design state
|
||||
(the §6 DesignObject serialization), so a future import reconstructs the
|
||||
DesignObject exactly.
|
||||
|
||||
Plus a stable **`design_id`** (the durable library key, ≠ Worldtree
|
||||
`session_id`) and a **`schema_version`**, both at the top level.
|
||||
|
||||
**Frame Invariant 1 is preserved — and now holds literally.** `ship.native` is a
|
||||
valid Worldtree Tier-3 `agents.define` payload assembled from `agent_name` + the
|
||||
designed agent's **`role`** (the model-role, resolved below) + the AUTHORED
|
||||
`system_prompt` (INV-E2-2 — never `composed_preview`) + `persona.ocean`
|
||||
(Worldtree renders affect at runtime) + `persona.psychological_profile` (the
|
||||
native home, LOCKED b53 per the vendored spec §4) + `motivational` (from
|
||||
goals_fears). The image and tools are NOT in the native schema — they ride the
|
||||
sidecar (tools bind via Bifrost at session-create, exactly as grounded).
|
||||
|
||||
**The `role` resolution (operator ruling 2026-07-13).** The blast-radius pass
|
||||
caught that `agents.define` requires `role` (a model-role slug, ADR-0012) but the
|
||||
design had no source for it. Resolution: **role is a first-class design field**,
|
||||
set by a new E3a **`set_role`** tool from a **curated 4-value enum** —
|
||||
`assistant` (general LLM), `thoughtful-assistant` (CoT general),
|
||||
`character` (RP/writing-tuned), `thoughtful-character` (CoT RP). This mirrors the
|
||||
D2 curated-style-modes decision: a fixed semantic set the operator picks from,
|
||||
NOT a coupling to any one deployment's arbitrary role registry. Export emits
|
||||
`design.role`, so the native payload is directly POST-valid (modulo the tool
|
||||
binding every consumer already supplies at session-create). The one deploy-time
|
||||
caveat: the 4 slugs must be granted on the target Worldtree (open_question B).
|
||||
|
||||
**The psych field is RESOLVED (no longer quarantined).** Vendored spec §4 locks
|
||||
`persona.psychological_profile` (prose `str`, ~150–300 words, read every turn),
|
||||
nesting under the `Any`-typed persona layer. Export maps `design.psych_profile`
|
||||
there and NOWHERE else — spec §4's hard constraint is that the self-report lens
|
||||
reads ONLY this field (leaking psych prose into `behavioral_notes`/`system_prompt`
|
||||
causes the "executive-assistant" failure).
|
||||
|
||||
## Data flow
|
||||
|
||||
**In:** a `DesignObject` (final, from E3) + a caller-supplied `design_id` (str)
|
||||
+ an optional caller-supplied `exported_at` (str | None). **Out:** a plain
|
||||
JSON-ready `dict` — the versioned bundle. **On disk / network:** NONE. Export is
|
||||
pure: the OCEAN parity gate (`validate_ocean`), the role-enum gate
|
||||
(`validate_role`), and the export-critical validators are in-memory; timestamps +
|
||||
ids come in as params; no clock, no randomness, no file, no HTTP. (Library
|
||||
persistence writes the returned dict to the JSON dir — that is the library epic,
|
||||
not this module.)
|
||||
|
||||
### Export bundle schema (v1.0)
|
||||
|
||||
```
|
||||
{
|
||||
"schema_version": "1.0", # ALWAYS EXPORT_SCHEMA_VERSION — not a caller param
|
||||
"design_id": "<caller-supplied durable library key, ≠ WT session_id>",
|
||||
"exported_at": <caller-supplied OPAQUE str | null — conventionally ISO-8601, NOT validated by export>,
|
||||
"ship": {
|
||||
"native": { # a valid agents.define payload (Frame Invariant 1)
|
||||
"agent_name": <str, non-blank, ≤128>,
|
||||
"role": <one of ROLE_CHOICES: assistant|thoughtful-assistant|character|thoughtful-character>,
|
||||
"system_prompt": <str, non-blank, ≤32768 — the AUTHORED block, INV-E2-2>,
|
||||
"persona": {
|
||||
"ocean": {O,C,E,A,N}, # each a real number in [-1,1] (validate_ocean parity)
|
||||
"psychological_profile": <str> # persona.psychological_profile (LOCKED b53); included iff non-blank
|
||||
},
|
||||
"motivational": {"goals": [...], "fears": [...]} # included iff goals_fears present + non-empty
|
||||
},
|
||||
"sidecar": {
|
||||
"portrait": <image ref str | null>, # only when portrait.status == "ready"; E4 owns generation
|
||||
"tools": [{"id","name","description"}],# the Bifrost tool manifest (bind at session-create)
|
||||
"first_message": <str> # the D3 opening turn (issue #347 seed)
|
||||
}
|
||||
},
|
||||
"resume": { <the §6 camelCase editable state — key set inlined below> }
|
||||
}
|
||||
```
|
||||
|
||||
**The `resume` key set (inlined — heid-review fold Gróa #9).** The resume half IS
|
||||
`serialize_design(design)` (relocated to `soong_lab.design`, R1), but its key set is
|
||||
pinned HERE so this contract is self-contained and an implementer knows the exact
|
||||
round-trip surface without reading the external, being-relocated function:
|
||||
|
||||
```
|
||||
resume = {
|
||||
"agentName", "role", "systemPrompt", "composedPreview", "firstMessage",
|
||||
"ocean" {O,C,E,A,N}, "dispositionPhrase", "psychProfile",
|
||||
"tools" [{id,name,description}], "portrait" {status, styleMode, imageUrl?, jobId?},
|
||||
"goalsFears" {goals,fears} | null
|
||||
}
|
||||
```
|
||||
|
||||
Import reconstructs a DesignObject from exactly these keys. `role` (new, R1) MUST be
|
||||
present so a reopened design carries its model-role. (`composedPreview` +
|
||||
`dispositionPhrase` are design-time-derived and re-derivable, but they ride the resume
|
||||
so a reopen renders instantly before the first recompute.)
|
||||
|
||||
**Divergences from the imported web mock (settled here, they were UI-comp
|
||||
shortcuts):**
|
||||
|
||||
| Field | Mock (web/*.js) | Real export (this contract) |
|
||||
|---|---|---|
|
||||
| native shape | `{name, tier, system_prompt, personality:{model,values}}` | real `agents.define` (`agent_name`/`role`/`persona.ocean`/`motivational`) |
|
||||
| role | absent | `design.role` ∈ ROLE_CHOICES |
|
||||
| system_prompt | `composedPreview` (mockApi) | authored `system_prompt` (INV-E2-2) |
|
||||
| psychProfile | omitted ("open backend decision") | `persona.psychological_profile` (LOCKED b53) |
|
||||
| bundle identity | none | `design_id` + `schema_version` |
|
||||
| resume half | none | full `serialize_design` state |
|
||||
|
||||
## Invariants
|
||||
|
||||
- **INV-E5-1** [hard]: `ship.native` is a valid Worldtree `agents.define` payload
|
||||
MODULO the tool binding — it carries every required field (`agent_name`,
|
||||
`role`, `system_prompt`) + `persona.ocean`, and OMITS only the tools (they bind
|
||||
via Bifrost at session-create, as they already do). Any `persona.ocean` export
|
||||
emits passes `validate_ocean`; `role` is always one of ROLE_CHOICES.
|
||||
`persona.psychological_profile` + `motivational` are OPTIONAL native fields
|
||||
(grounded) — omitting them when blank/empty keeps the payload fully valid, not
|
||||
merely "valid enough" (heid-review fold, Gróa #1).
|
||||
- **INV-E5-2** [hard]: The exported `system_prompt` is the AUTHORED
|
||||
`design.system_prompt`, NEVER `composed_preview` (binds with INV-E2-2). The
|
||||
**disposition line** — the `"Disposition: <name> is <phrase>."` sentence that
|
||||
E2 `recompute` appends to `composed_preview` (design_object.contract.md POST-E2-5)
|
||||
— is design-time-only and never ships.
|
||||
- **INV-E5-3** [hard]: Export is pure + deterministic — identical
|
||||
`(design, design_id, exported_at)` inputs yield a byte-identical serialized
|
||||
bundle. No clock, no randomness, no I/O. The determinism is WITHIN the module:
|
||||
the returned dict has a fixed key insertion order (schema_version, design_id,
|
||||
exported_at, ship, resume; native + sidecar likewise), so any consistent
|
||||
`json.dumps` settings produce byte-identical output — the invariant does NOT
|
||||
claim cross-implementation byte-identity (heid-review fold, Regin #6).
|
||||
- **INV-E5-4** [hard]: Export NEVER mutates the input `DesignObject` (read-only);
|
||||
the bundle holds copies, not aliases, of every mutable sub-structure (ocean
|
||||
dict, tool list, goals/fears lists) so a later design mutation can't change an
|
||||
already-built bundle.
|
||||
- **INV-E5-5** [hard]: `validate_exportable` is the strict export-critical gate
|
||||
(decision #6): OCEAN (via `validate_ocean`), role (∈ ROLE_CHOICES via
|
||||
`validate_role`), agent_name (non-blank, ≤128), system_prompt (non-blank,
|
||||
≤32768), tool-refs (id/name non-blank + bounded). A design that fails ANY of
|
||||
these raises `ExportError` and NO bundle is produced — a built bundle is always
|
||||
well-formed enough to round-trip on import.
|
||||
- **INV-E5-6** [hard]: The bundle carries `schema_version` at the top level, and
|
||||
readers (import, future) MUST tolerate unknown extra keys (lenient on unknown
|
||||
metadata, decision #6) — the schema is add-only-friendly.
|
||||
- **INV-E5-7** [hard]: `psych_profile` maps to `persona.psychological_profile`
|
||||
and NOWHERE else — it never leaks into `behavioral_notes`, `system_prompt`, or
|
||||
any other native field (vendored spec §4 hard constraint — the lens reads only
|
||||
this dedicated field).
|
||||
|
||||
## Constraints
|
||||
|
||||
- **[correctness]** `validate_exportable`'s OCEAN check IS `validate_ocean` and
|
||||
its role check IS `validate_role` (both E2) — no re-implementation, no drift.
|
||||
The LENGTH bounds (name, prompt, tool id/name/desc, psych_profile, first_message)
|
||||
MUST equal the E3a tool-schema caps — now shared constants in `soong_lab.design`
|
||||
(`AGENT_NAME_MAX`, `SYSTEM_PROMPT_MAX`, `PSYCH_PROFILE_MAX`, `FIRST_MESSAGE_MAX`,
|
||||
`TOOL_*_MAX`), imported by BOTH bifrost/tools.py and export — so a design's field
|
||||
LENGTHS never drift. Import the shared constants; do not re-declare the numbers.
|
||||
(Export is stricter only on whitespace-blankness of the required fields — the one
|
||||
intentional one-way difference from the tools' minLength:1.)
|
||||
- **[style]** Pure — NO I/O (no clock, no file, no HTTP, no randomness). Every
|
||||
time-varying value (`design_id`, `exported_at`) is a param.
|
||||
- **[explicit]** The one deploy-time caveat (the 4 role slugs must be granted on
|
||||
the target WT) is documented in THIS contract (open_question B) + the library /
|
||||
README when it lands — NOT promised as a bundle/sidecar field (heid-review fold:
|
||||
the bundle is machine-consumed; a human deploy-note is not bundle data). The
|
||||
bundle carries the `role` value; slug-grant validity is a deploy concern.
|
||||
- **[explicit]** `build_export_bundle` is the PUBLIC entrypoint — it runs the
|
||||
validate→assemble ordering. `build_native_payload` / `build_sidecar` are exposed
|
||||
for testing + reuse but ASSUME an already-validated design (PRE-E5-2 / PRE-E5-4);
|
||||
a direct caller that skips `validate_exportable` owns that gate (heid-review fold,
|
||||
Hulda #5).
|
||||
|
||||
```contract
|
||||
FN validate_exportable(design: DesignObject) -> None
|
||||
BRIEF: The strict export-critical gate (settled decision #6) — refuse to build a bundle from a design that would fail on re-import or at the designed agent's define/session-create. Checks OCEAN (validate_ocean), role (validate_role), agent_name, system_prompt, every tool-ref, and the psych_profile/first_message LENGTH — against the SAME length caps the E3a tools enforce (shared constants). NO-DRIFT is one-directional: export's LENGTH bounds equal the tool caps, but export is deliberately STRICTER on whitespace — a whitespace-only required field (name/prompt/tool id/name) passes the tools' minLength:1 yet is rejected here (a " " name must not ship). Raises ExportError with the offending field; never mutates the design.
|
||||
PRE: [PRE-E5-1 hard] design is a DesignObject
|
||||
POST: [POST-E5-1 exception] raises ExportError(field, detail) unless ALL hold: design.ocean passes validate_ocean; design.role passes validate_role (∈ ROLE_CHOICES); agent_name is a non-blank str of len ≤ _AGENT_NAME_MAX; system_prompt is a non-blank str of len ≤ _SYSTEM_PROMPT_MAX; every tool has non-blank str id (≤_TOOL_ID_MAX) + non-blank str name (≤_TOOL_NAME_MAX) + str description (≤_TOOL_DESC_MAX); psych_profile is a str of len ≤ _PSYCH_PROFILE_MAX (blank OK); first_message is a str of len ≤ _FIRST_MESSAGE_MAX (blank OK). The id/name-required vs description/psych/first_message-may-be-blank asymmetry is INTENTIONAL — description defaults to "" via attach_tool; psych_profile/first_message are optional prose so only their LENGTH is bounded, not blankness (heid-review Gróa #8 + correctness-finder folds)
|
||||
POST: [POST-E5-2 state_change] design is unchanged — no mutation (INV-E5-4)
|
||||
STEPS:
|
||||
1. [setup, flexibility=prescriptive] TRY validate_ocean(design.ocean) — on OceanError, RAISE ExportError("persona.ocean", str(exc)) (reuse E2, no re-impl)
|
||||
2. [sequential, flexibility=prescriptive] TRY validate_role(design.role) — on RoleError, RAISE ExportError("role", str(exc)) (reuse E2 role canon)
|
||||
3. [branch] IF agent_name is not a non-blank str OR len > _AGENT_NAME_MAX: RAISE ExportError("agent_name", ...)
|
||||
4. [branch] IF system_prompt is not a non-blank str OR len > _SYSTEM_PROMPT_MAX: RAISE ExportError("system_prompt", ...) # the AUTHORED block, INV-E5-2
|
||||
5. [loop] FOR EACH tool in design.tools: IF id/name blank or over max, or description non-str/over max: RAISE ExportError(f"tools[{i}]", ...)
|
||||
6. [branch] IF psych_profile is non-str OR len > _PSYCH_PROFILE_MAX: RAISE ExportError("psych_profile", ...) # length only — blank OK (optional prose)
|
||||
7. [branch] IF first_message is non-str OR len > _FIRST_MESSAGE_MAX: RAISE ExportError("first_message", ...) # length only — blank OK
|
||||
8. [cleanup] RETURN None
|
||||
TESTS:
|
||||
minimal_ok [happy,tracer]: agent_name+system_prompt set, role="character", neutral OCEAN, no tools → no raise
|
||||
blank_name [adversarial]: agent_name="" → ExportError("agent_name")
|
||||
blank_prompt [adversarial]: system_prompt=" " → ExportError("system_prompt")
|
||||
prompt_too_long [boundary]: system_prompt of len _SYSTEM_PROMPT_MAX+1 → ExportError; len _SYSTEM_PROMPT_MAX → ok
|
||||
bad_ocean [adversarial]: ocean missing a key → ExportError("persona.ocean") (via validate_ocean)
|
||||
bad_role [adversarial]: role="wizard" (not in ROLE_CHOICES) → ExportError("role") (via validate_role)
|
||||
blank_role [adversarial]: role="" → ExportError("role")
|
||||
bad_tool_ref [adversarial]: a tool with id="" → ExportError("tools[0]")
|
||||
no_mutation [property]: a rejected design is byte-identical before/after the raise (INV-E5-4)
|
||||
psych_profile_length [boundary]: psych_profile="" → ok; len _PSYCH_PROFILE_MAX+1 → ExportError("psych_profile")
|
||||
first_message_length [boundary]: first_message len _FIRST_MESSAGE_MAX+1 → ExportError("first_message"); blank → ok
|
||||
whitespace_name_rejected [adversarial]: agent_name=" " → ExportError("agent_name") — deliberately stricter than the tool's minLength:1 (a whitespace-only name must not ship)
|
||||
length_bounds_parity [property]: any (name, prompt, tool, psych, first_message) LENGTH the E3a tool schema accepts is ≤ export's caps (shared constants); export is stricter ONLY on whitespace-blankness of required fields, never looser on length
|
||||
```
|
||||
|
||||
```contract
|
||||
FN build_native_payload(design: DesignObject) -> dict[str, Any]
|
||||
BRIEF: Map a DesignObject to a valid native agents.define payload (Frame Invariant 1). Emits agent_name + role + the AUTHORED system_prompt + persona{ocean, psychological_profile?} + motivational?. Copies mutable sub-structures (INV-E5-4). Assumes validate_exportable already passed (called by build_export_bundle).
|
||||
PRE: [PRE-E5-2 hard] design passed validate_exportable (OCEAN valid, role valid, name/prompt present) — build_export_bundle enforces this ordering
|
||||
POST: [POST-E5-3 return_value] result has agent_name == design.agent_name, role == design.role (∈ ROLE_CHOICES), and system_prompt == design.system_prompt (the AUTHORED block, INV-E5-2), and result["persona"]["ocean"] == a COPY of design.ocean
|
||||
POST: [POST-E5-4 return_value] result["role"] == design.role — the designed agent's model-role (one of the 4 ROLE_CHOICES); a valid agents.define required field
|
||||
POST: [POST-E5-5 return_value] persona.psychological_profile == design.psych_profile when psych_profile is non-blank, else the key is absent; it appears under persona and NOWHERE else (INV-E5-7)
|
||||
POST: [POST-E5-6 return_value] motivational == {"goals": copy, "fears": copy} when design.goals_fears is present AND at least one list is non-empty; else the key is absent (never an empty motivational block)
|
||||
STEPS:
|
||||
1. [setup] payload = {"agent_name": design.agent_name, "role": design.role, "system_prompt": design.system_prompt} # role emitted; system_prompt is the authored block (INV-E5-2)
|
||||
2. [sequential] persona = {"ocean": dict(design.ocean)} # COPY, not alias (INV-E5-4)
|
||||
3. [branch] IF design.psych_profile is a non-blank str: persona["psychological_profile"] = design.psych_profile # LOCKED b53 field; ONLY here (INV-E5-7)
|
||||
4. [sequential] payload["persona"] = persona
|
||||
5. [branch] IF design.goals_fears is not None AND (goals or fears non-empty): payload["motivational"] = {"goals": list(gf.goals), "fears": list(gf.fears)}
|
||||
6. [cleanup] RETURN payload # tools NOT here — they ride the sidecar / bind via Bifrost at session-create
|
||||
TESTS:
|
||||
authored_prompt [happy,tracer]: system_prompt authored + composed_preview differs → payload.system_prompt == authored, NOT composed_preview (INV-E5-2)
|
||||
role_emitted [happy]: role="thoughtful-character" → payload.role == "thoughtful-character" (POST-E5-4)
|
||||
ocean_copied [property]: mutate design.ocean after build → payload's ocean unchanged (INV-E5-4)
|
||||
psych_present [happy]: psych_profile set → persona.psychological_profile == it; it is the ONLY field carrying it (INV-E5-7)
|
||||
psych_absent [boundary]: psych_profile="" → no psychological_profile key
|
||||
motivational_present [happy]: goals_fears with goals=["x"] → motivational.goals == ["x"]
|
||||
motivational_absent [boundary]: goals_fears None → no motivational key; goals_fears with both lists empty → no motivational key
|
||||
no_tools_no_image [trace]: payload has no "tools" and no image field (they ride the sidecar / bind separately)
|
||||
```
|
||||
|
||||
```contract
|
||||
FN build_sidecar(design: DesignObject) -> dict[str, Any]
|
||||
BRIEF: Assemble the soong-lab sidecar — the three artifacts the native schema has no home for: the persona portrait ref, the Bifrost tool manifest, and the D3 first_message. Copies the tool list (INV-E5-4).
|
||||
PRE: [PRE-E5-4 hard] design is a DesignObject (its portrait/tools/first_message fields are read as-is; no validation here — validate_exportable is the gate, called by build_export_bundle before this)
|
||||
POST: [POST-E5-7 return_value] result == {"portrait": <str|None>, "tools": [{"id","name","description"} per tool, copied], "first_message": design.first_message}; portrait == design.portrait.image_url IFF design.portrait.status == "ready", else None (a "ready" status with a None image_url therefore yields None — no crash; any non-"ready" status → None — heid-review fold Gróa #4)
|
||||
STEPS:
|
||||
1. [setup] portrait = design.portrait.image_url if design.portrait.status == "ready" else None
|
||||
2. [sequential] tools = [t.to_dict() for t in design.tools] # ToolRef.to_dict() — the shared {id,name,description} projection (dedups with serialize_design); it MUST emit exactly id/name/description, so if to_dict ever grows keys the sidecar spec must be revisited (heid-code-review fold)
|
||||
3. [cleanup] RETURN {"portrait": portrait, "tools": tools, "first_message": design.first_message}
|
||||
TESTS:
|
||||
ready_portrait [happy]: portrait.status="ready", image_url set → sidecar.portrait == the url
|
||||
unready_portrait [boundary]: portrait.status="generating" (url set) → sidecar.portrait is None (only ready ships)
|
||||
none_portrait [boundary]: portrait.status="none" → sidecar.portrait is None
|
||||
tools_manifest [happy,tracer]: two tools → sidecar.tools has both {id,name,description}
|
||||
tools_copied [property]: mutate design.tools after build → sidecar.tools unchanged (INV-E5-4)
|
||||
first_message [happy]: first_message set → sidecar.first_message == it
|
||||
```
|
||||
|
||||
```contract
|
||||
FN build_export_bundle(design: DesignObject, *, design_id: str, exported_at: str | None = None) -> dict[str, Any]
|
||||
BRIEF: The top-level export entrypoint — validate (strict, INV-E5-5), then assemble the versioned bundle: {schema_version, design_id, exported_at, ship:{native, sidecar}, resume}. Pure + deterministic (INV-E5-3); the caller supplies design_id + exported_at (no clock here). The resume half reuses serialize_design (the §6 state) so import round-trips. schema_version is NOT a caller param (heid-review fold) — it is ALWAYS EXPORT_SCHEMA_VERSION, so a bundle's version is never caller-forgeable; a future migration bumps the module constant. exported_at is an OPAQUE caller-supplied string (conventionally ISO-8601) — export does NOT parse or validate it (purity; the caller owns timestamp correctness).
|
||||
PRE: [PRE-E5-3 hard] design_id is a non-blank str (the durable library key) — a blank id RAISES ExportError("design_id", ...) (a bundle with no library key is unusable)
|
||||
POST: [POST-E5-8 exception] IF the design fails validate_exportable, the ExportError propagates and NO bundle is returned (INV-E5-5) — validation is BEFORE assembly
|
||||
POST: [POST-E5-9 return_value] returns {schema_version: EXPORT_SCHEMA_VERSION (always), design_id, exported_at, ship:{native: build_native_payload(design), sidecar: build_sidecar(design)}, resume: serialize_design(design)}; exported_at is the param verbatim (None → JSON null), unvalidated
|
||||
POST: [POST-E5-10 return_value] deterministic — identical (design, design_id, exported_at) → byte-identical json.dumps(result) given fixed dumps settings; the returned dict has a FIXED key insertion order (schema_version, design_id, exported_at, ship, resume), so a caller's json.dumps is stable (INV-E5-3); design unchanged (INV-E5-4)
|
||||
STEPS:
|
||||
1. [setup, flexibility=prescriptive] IF design_id is not a non-blank str: RAISE ExportError("design_id", "a non-blank design_id is required")
|
||||
2. [sequential] CALL validate_exportable(design) # strict gate BEFORE assembly (INV-E5-5) — raises propagate
|
||||
3. [sequential] native = build_native_payload(design); sidecar = build_sidecar(design); resume = serialize_design(design)
|
||||
4. [cleanup] RETURN {"schema_version": EXPORT_SCHEMA_VERSION, "design_id": design_id, "exported_at": exported_at, "ship": {"native": native, "sidecar": sidecar}, "resume": resume}
|
||||
TESTS:
|
||||
full_bundle [happy,tracer]: a complete design + design_id="d-1" → bundle has schema_version, design_id=="d-1", ship.native.agent_name, ship.native.role, ship.sidecar.first_message, resume.systemPrompt
|
||||
blank_design_id [adversarial]: design_id="" → ExportError("design_id") before any assembly
|
||||
invalid_design_no_bundle [adversarial]: a design with blank agent_name → ExportError propagates, no dict returned (POST-E5-8)
|
||||
deterministic [property]: build twice with the same (design, design_id, exported_at) → byte-identical json.dumps (INV-E5-3)
|
||||
exported_at_passthrough [trace]: exported_at="2026-07-13T00:00:00Z" → bundle.exported_at == it verbatim; None → null; a non-ISO "banana" is passed through unvalidated
|
||||
schema_version_not_a_param [trace]: build_export_bundle(..., schema_version="banana") raises TypeError — schema_version is fixed, never caller-supplied (heid-review fold)
|
||||
resume_roundtrips [property]: resume half == serialize_design(design) — every editable field present for import (incl. role)
|
||||
no_mutation [property]: design byte-identical before/after build (INV-E5-4)
|
||||
schema_version_present [trace]: bundle.schema_version == EXPORT_SCHEMA_VERSION (INV-E5-6)
|
||||
```
|
||||
|
||||
## Integration points
|
||||
|
||||
**R1 — relocate `serialize_design` out of `web.py` (agent-discretion refactor,
|
||||
no public-surface change).** The resume half reuses the §6 DesignObject
|
||||
serialization, but `serialize_design` currently lives in `soong_lab.web`
|
||||
(Starlette-coupled). Importing `web.py` into `export` would drag Starlette +
|
||||
the orchestrator into a pure module. Fix: **move `serialize_design` to
|
||||
`soong_lab.design`** (it is a pure `DesignObject → dict` mapping with no web
|
||||
dependency — it belongs with the model; add `role` to its output), and update the
|
||||
two consumers to import it from there. Blast radius (confirmed via grep):
|
||||
`web.py` (define → import; 3 call-sites unchanged), `tests/test_web.py:23`
|
||||
(import path), and the new `export` consumer. Behavior-identical;
|
||||
`web_surface.contract.md` gets a one-line note. No-backwards-compat: the old
|
||||
location is deleted, all refs updated in the same commit.
|
||||
|
||||
**Companion prerequisite slice — the `role` field + `set_role` tool (contracts
|
||||
updated in THIS pass).** Export emits `design.role`, so the field + its tool must
|
||||
exist. This slice (governed by the sibling contracts, amended alongside this one):
|
||||
- `soong_lab.design` (design_object.contract.md): a `role` field on
|
||||
`DesignObject` (default `"character"`); a `ROLE_CHOICES` enum canon +
|
||||
`validate_role`, held as an in-code module constant (mirroring the OCEAN
|
||||
adjective canon); `new_design()` sets `role="character"`; `serialize_design`
|
||||
adds `role`.
|
||||
- `soong_lab.bifrost` (bifrost_server.contract.md): a new `set_role(_ctx, role)`
|
||||
design tool (the 9th), `input_schema` an `enum` of the 4 values; the handler
|
||||
sets `design.role` after membership validation.
|
||||
The behavioral CODE for this slice lands in the TDD phase after
|
||||
`/heid-contract-review`, alongside `soong_lab.export`.
|
||||
|
||||
**Bifrost export tool (`_make_export` in bifrost/tools.py) — in scope.** Replace
|
||||
the deferred stub with: get the session's design from the store, then
|
||||
`build_export_bundle(design, design_id=<source>, exported_at=<stamp>)` and
|
||||
return the bundle (or a compact confirmation carrying it). The `design_id`
|
||||
source is the per-design-sessions seam (open_question C) — until it lands, the
|
||||
tool may pass the session_id as a provisional design_id (a documented
|
||||
placeholder, NOT a silent default). The tool handler is the impure boundary that
|
||||
stamps `exported_at` (clock) and supplies `design_id`, keeping
|
||||
`soong_lab.export` pure.
|
||||
|
||||
**`/api/export` endpoint + web/api.js shim — NOT in this contract (open_question
|
||||
D).** The browser 'Export Asset' button calls `api.export()`, today a
|
||||
client-side shim assembling a NON-native mock bundle. The real path is a thin
|
||||
`GET /api/export` on `web.py` → `build_export_bundle(orchestrator.get_design(),
|
||||
…)` → JSON → the modal's native/sidecar panes render it. That amends
|
||||
`web_surface.contract.md`; it is a follow-up slice in the same epic, specified
|
||||
here only so the seam is visible.
|
||||
|
||||
## Downstream epics (NOT this contract)
|
||||
|
||||
- **Library persistence** (decision #5) — writing the returned bundle to the
|
||||
server-local single-user JSON dir on corviduo-dev, keyed by `design_id`; the
|
||||
minimal recent-designs picker.
|
||||
- **Import** (decision #6) — reading a bundle: lenient on unknown metadata
|
||||
(INV-E5-6), STRICT re-validation of the export-critical fields (the import-side
|
||||
mirror of `validate_exportable`), reconstructing a DesignObject from the
|
||||
`resume` half.
|
||||
- **Per-design-sessions** (decision #2) — the `design_id` generator + the
|
||||
fresh-WT-session-per-open lifecycle (also caps the #355 accumulation).
|
||||
+384
@@ -0,0 +1,384 @@
|
||||
---
|
||||
contract_version: "2.1"
|
||||
module: "soong_lab.importer"
|
||||
purpose: "Reconstruct a DesignObject from an export bundle's `resume` half — the inverse of soong_lab.export. HYBRID validation (settled decision #6): LENIENT on unknown metadata (unknown top-level bundle keys, unknown keys inside resume, any schema_version), STRICT re-validation of the export-critical fields (OCEAN, role ∈ ROLE_CHOICES, agent_name, system_prompt length, tool-refs, psych/first_message length) surfaced ON IMPORT so a truncated or tampered bundle fails EARLY, not after more tuning. Pure + deterministic: no I/O, no persistence, no network, no clock (library read + the /api/import endpoint + the reopen lifecycle are separate downstream epics)."
|
||||
depends_on:
|
||||
- "soong_lab.design" # DesignObject/ToolRef/Portrait/GoalsFears + serialize_design (the round-trip partner) + ROLE_CHOICES/UNSET_ROLE + the shared field-bound constants
|
||||
- "soong_lab.export" # validate_exportable + ExportError — the strict export-critical gate is REUSED, not re-implemented (no-drift, INV-I-1)
|
||||
used_by:
|
||||
- "soong_lab.web" # FUTURE (import epic) — the POST /api/import endpoint parses the uploaded bundle JSON → import_bundle → seed a session (out of scope here, open_question D)
|
||||
- "soong_lab.soong" # FUTURE (per-design-sessions) — the reopen lifecycle imports a stored bundle, opens a fresh WT session, seeds the design-state summary (out of scope, decision #2)
|
||||
language: "python"
|
||||
complexity: "medium"
|
||||
estimated_loc: 170
|
||||
confidence: 0.83
|
||||
assumptions:
|
||||
- "Import consumes a Python dict (a Mapping), NOT raw bytes/JSON text. The JSON parse (json.loads at the /api/import endpoint or the library-read layer) happens UPSTREAM; import operates on the already-parsed structure, exactly as export RETURNS a Python dict the caller json.dumps'es. So the round-trip contract is over Python dicts: import_bundle(build_export_bundle(d, design_id=…)) == d, with no JSON layer in between (the JSON boundary — float/int coercion, encoding — is the endpoint/library epic's concern, INV-I-5 note)."
|
||||
- "The `resume` half is the ONLY source of truth on import (settled decision #4 — resume is 'what you reopen to keep tuning'). The `ship` half is a re-derivable deployment artifact; import IGNORES it. The reopen path re-exports from the reconstructed design, regenerating ship, so a ship↔resume mismatch is harmless — resume wins (INV-I-5). No cross-check in v1."
|
||||
- "The export-critical gate on import IS soong_lab.export.validate_exportable, imported and reused verbatim — NOT a re-implemented import-side validator. This guarantees import can never drift looser than export: the exact fields export refuses to ship are the exact fields import refuses to accept (INV-I-1). ExportError is caught and re-raised as BundleImportError so callers get an import-shaped error while the validation authority stays single-sourced."
|
||||
- "role is a first-class DesignObject field (operator ruling 2026-07-13), one of the curated ROLE_CHOICES, set by the E3a set_role tool. A resume carries `role`; import restores it and validate_role (via validate_exportable) rejects UNSET_ROLE ('') or any non-member — you cannot re-import an unclassified design, same as you cannot export one."
|
||||
- "composed_preview + disposition_phrase ride the resume so a reopen renders instantly (export.contract §resume). Import TRUSTS these verbatim (INV-I-8) — it does NOT call recompute. Re-derivation from ocean+prompt is the reopen lifecycle's concern (per-design-sessions), not import's. For a legitimately-exported bundle they are already self-consistent; a hand-tampered preview is design-time-only and is overwritten on the next set_ocean/edit_prompt recompute."
|
||||
open_questions:
|
||||
- "[A — RESOLVED, operator 2026-07-13] Module name is `soong_lab.importer` (operator chose it over `soong_lab.ingest`; keyword-safe agent-noun mirroring `export`). The export contract's forward-reference `used_by: soong_lab.import` — an unusable Python-keyword path (`import soong_lab.import` is a SyntaxError) — is corrected to `soong_lab.importer` in the same commit (done). SETTLED: the Constraints hard-require reflects the decision, not a still-open recommendation (heid-review Gróa#1 reconcile open-vs-locked)."
|
||||
- "[B — SETTLED, agent-discretion] Error type is `BundleImportError(field, detail)`, mirroring export's `ExportError(field, detail)`. Deliberately NOT `ImportError` — that shadows the Python builtin, a foot-gun for an import module. The Constraints hard-require reflects the decision, not a still-open recommendation (heid-review Gróa#1)."
|
||||
- "[C — presence vs default, agent-discretion, notable] For the export-critical resume keys (agentName, role, systemPrompt, ocean) a MISSING key is a hard reject (INV-I-7), NOT a silent default. Rationale: a missing `ocean` would default to a VALID neutral OCEAN and pass validate_exportable — silently masking trait loss from a truncated bundle. Rejecting on absence fails loudly + consistently (the 'fail early on import' the decision wants). Rejected alternative: reconstruct-with-defaults-then-validate (inconsistent — ocean slips through while name/role are caught by validation)."
|
||||
- "[D — scope] This contract = the PURE reconstruction (deserialize_design) + the strict entrypoint (import_bundle) + BundleImportError. The POST /api/import endpoint (amends web_surface.contract.md), the reopen Bifrost tool / session-open wiring (per-design-sessions), and reading a bundle off the library JSON dir (library epic) are ALL downstream — specified here only as the integration seam so it is visible. Nothing in this contract does I/O."
|
||||
- "[E — schema_version tolerance] `schema_version` is read at the bundle TOP LEVEL only (where export stamps EXPORT_SCHEMA_VERSION) — import does not look for it inside `resume`. v1 tolerates ANY top-level value (present or absent) and reads the v1 resume key set regardless (INV-I-2, INV-E5-6 add-only-friendly). 'Tolerate any version' means forward-compat with ADD-ONLY future changes — NOT a promise of semantic compatibility with a bundle whose meaning changed (heid-review Gróa#5/Hulda). A future policy — reject an incompatible MAJOR version, or dispatch to a version-specific deserializer — is deferred to when a second schema version actually exists. v1 has exactly one shape."
|
||||
---
|
||||
|
||||
## Context
|
||||
|
||||
Import is the SECOND half of the operator-accepted (2026-07-13) export/import/library
|
||||
design — the block that expands the locked single-agent frame into a multi-pass
|
||||
tuning loop (design → export → **reopen → tune** → keep a library). Where
|
||||
`soong_lab.export` turns a finished `DesignObject` into a versioned bundle, this
|
||||
module does the inverse: it takes a bundle's **`resume`** half and reconstructs an
|
||||
editable `DesignObject` you can drop back into a session and keep tuning.
|
||||
|
||||
The reconstruction is **HYBRID-validated** (settled decision #6 — the load-bearing
|
||||
import decision):
|
||||
|
||||
- **LENIENT on unknown metadata.** Unknown top-level bundle keys, unknown keys
|
||||
inside `resume`, and any `schema_version` (present or absent) are tolerated —
|
||||
import reads only the keys it knows (INV-I-2, mirroring the export bundle's
|
||||
add-only-friendly `INV-E5-6`). A bundle from a future soong-lab that added
|
||||
fields still imports.
|
||||
- **STRICT on the export-critical fields.** OCEAN, `role`, `agent_name`,
|
||||
`system_prompt`, tool-refs, and the psych/first_message length are re-validated
|
||||
**on import** by REUSING `soong_lab.export.validate_exportable` verbatim (INV-I-1)
|
||||
— so the exact fields export refuses to *ship* are the exact fields import
|
||||
refuses to *accept*, and import can never drift looser than export. A bad field
|
||||
is surfaced immediately (fail EARLY), not after the operator has tuned for
|
||||
another ten minutes against a design that was never valid.
|
||||
|
||||
**The round-trip is the load-bearing contract between the two modules** (INV-I-3):
|
||||
for any exportable design `d`,
|
||||
|
||||
```
|
||||
import_bundle(build_export_bundle(d, design_id="…")) == d
|
||||
deserialize_design(serialize_design(d)) == d
|
||||
```
|
||||
|
||||
This is what makes "export then reopen" lossless. `serialize_design`
|
||||
(relocated to `soong_lab.design` in the export pass, R1) is the forward half;
|
||||
`deserialize_design` here is its exact inverse.
|
||||
|
||||
**Import reads the `resume` half ONLY.** The `ship` half (native `agents.define`
|
||||
payload + sidecar) is a re-derivable deployment artifact — the reopen path
|
||||
re-exports from the reconstructed design, regenerating `ship`. So import ignores
|
||||
`ship` entirely (INV-I-5); a tampered `ship` that disagrees with `resume` is
|
||||
harmless (resume wins, ship regenerated). No cross-check in v1.
|
||||
|
||||
**What this contract does NOT do** (open_question D): no file read, no HTTP, no
|
||||
session seeding. The `POST /api/import` endpoint, the reopen Bifrost tool /
|
||||
session-open wiring, and reading a bundle off the library JSON dir are downstream
|
||||
epics. This module is the pure, side-effect-free reconstruction core those epics
|
||||
build on — exactly as `soong_lab.export` is the pure builder its endpoint wraps.
|
||||
|
||||
## Data flow
|
||||
|
||||
**In:** a bundle `dict` (a Mapping — already `json.loads`'d upstream). **Out:** a
|
||||
validated, ready-to-reopen `DesignObject`. **On disk / network:** NONE. Import is
|
||||
pure: the structural gate (bundle/resume/ocean are dicts, tools a list-of-dicts),
|
||||
the tolerant reconstruction, and the strict `validate_exportable` re-check are all
|
||||
in-memory; no clock, no randomness, no file, no HTTP.
|
||||
|
||||
### The resume key set consumed (v1.0)
|
||||
|
||||
Import reconstructs from exactly the `serialize_design` output (the §6 camelCase
|
||||
state — pinned in export.contract §resume, restated here so this contract is
|
||||
self-contained):
|
||||
|
||||
```
|
||||
resume = {
|
||||
"agentName": <str>, # EXPORT-CRITICAL — presence required (INV-I-7)
|
||||
"role": <str ∈ ROLE_CHOICES>, # EXPORT-CRITICAL — presence required; validate_role gates value
|
||||
"systemPrompt": <str>, # EXPORT-CRITICAL — presence required; the AUTHORED block
|
||||
"ocean": {O,C,E,A,N}, # EXPORT-CRITICAL — presence required; validate_ocean gates value
|
||||
"tools": [{id,name,description}], # optional (absent → []); each ref value-gated by validate_exportable
|
||||
"composedPreview": <str>, # design-time-derived — TRUSTED verbatim, re-derivable (INV-I-8)
|
||||
"dispositionPhrase": <str>, # design-time-derived — TRUSTED verbatim, re-derivable (INV-I-8)
|
||||
"firstMessage": <str>, # optional prose — length-gated only (blank OK)
|
||||
"psychProfile": <str>, # optional prose — length-gated only (blank OK)
|
||||
"portrait": {status, styleMode, imageUrl?, jobId?}, # optional (absent → default Portrait())
|
||||
"goalsFears": {goals,fears} | null # optional (absent/null → None)
|
||||
}
|
||||
```
|
||||
|
||||
**Critical vs optional (the presence rule, INV-I-7).** Read the two functions as a
|
||||
boundary (all three review arms flagged that the prose blurs it): the INNER
|
||||
`deserialize_design` is total and DEFAULTS every missing key (a missing `ocean` →
|
||||
neutral) — it NEVER rejects; the OUTER, public `import_bundle` PRESENCE-CHECKS the
|
||||
export-critical keys and REJECTS a missing one BEFORE it ever calls deserialize. So
|
||||
"import defaults a missing ocean to neutral" is FALSE for the public path
|
||||
(`import_bundle` rejects it, INV-I-7) — the neutral default lives ONLY inside the
|
||||
never-directly-shipped inner function (heid-review 3/3: POST-I-3 vs INV-I-7 read as
|
||||
contradictory in isolation). `agentName`, `role`, `systemPrompt`, `ocean` are
|
||||
**presence-required** — a missing one is a truncated / corrupt bundle and raises
|
||||
`BundleImportError`, because defaulting them would either be caught inconsistently
|
||||
(name/role/prompt default to values `validate_exportable` rejects) or silently
|
||||
masked (`ocean` defaults to a VALID neutral OCEAN — silent trait loss). Every other
|
||||
key is optional and defaults to the `DesignObject` default when absent. `tools`/`portrait`/`goalsFears`, when present, must be well-formed SHAPES —
|
||||
`tools` a list-of-objects, `ocean`/`portrait` an object, `goalsFears` null or an
|
||||
object whose present `goals`/`fears` are lists — structural mismatches raise a clean
|
||||
`BundleImportError`, never a leaked builtin `TypeError`/`ValueError` (INV-I-6
|
||||
robustness). These SHAPE gates all exist to prevent SILENT DATA LOSS (heid-bug-hunt
|
||||
Gróa#1/#2: a malformed portrait/goalsFears would otherwise coerce to a default in
|
||||
`deserialize_design` and slip PAST `validate_exportable`, since both are
|
||||
non-export-critical — the same loss the `tools` gate was added to close). Import does
|
||||
NOT validate their VALUE contents — portrait `status`/`styleMode` enums or goals/fears
|
||||
item contents are not export-critical (E4 / the UI own portrait validity); those
|
||||
round-trip as-is (heid-review Gróa#6).
|
||||
|
||||
## Invariants
|
||||
|
||||
- **INV-I-1** [hard]: The strict export-critical re-validation IS
|
||||
`soong_lab.export.validate_exportable`, imported and reused verbatim — NO
|
||||
re-implementation, no parallel import-side validator. Import therefore can NEVER
|
||||
be looser than export: OCEAN (`validate_ocean`), role (`validate_role`, ∈
|
||||
ROLE_CHOICES), `agent_name` (non-blank, ≤`AGENT_NAME_MAX`), `system_prompt`
|
||||
(non-blank, ≤`SYSTEM_PROMPT_MAX`), every tool-ref (id/name non-blank + bounded,
|
||||
description bounded), and the psych/first_message LENGTH are all gated by the
|
||||
same code export uses. An `ExportError` from that gate is caught and re-raised
|
||||
as `BundleImportError(same field, same detail)` — same field granularity,
|
||||
import-shaped type.
|
||||
- **INV-I-2** [hard]: LENIENT on unknown metadata (settled decision #6, mirrors
|
||||
INV-E5-6). Unknown top-level bundle keys, unknown keys inside `resume`, and any
|
||||
`schema_version` value (present or absent) are tolerated — import reads only the
|
||||
keys it knows and ignores the rest. A future-schema bundle that ADDED fields
|
||||
still imports.
|
||||
- **INV-I-3** [hard]: ROUND-TRIP — for any `DesignObject` `d` that passes
|
||||
`validate_exportable`, `deserialize_design(serialize_design(d))` reconstructs an
|
||||
EQUAL `DesignObject` (dataclass `==` over every field), and
|
||||
`import_bundle(build_export_bundle(d, design_id=…))` `== d`. This is the lossless
|
||||
export↔import contract. (Equality is over Python structures; the JSON encode/decode
|
||||
boundary is the endpoint/library epic's concern, not this module's.)
|
||||
- **INV-I-4** [hard]: NO-ALIAS — the reconstructed `DesignObject` holds COPIES of
|
||||
every mutable sub-structure (the ocean dict, the tools list, the goals/fears
|
||||
lists) drawn from the bundle, never aliases. A later mutation of the input bundle
|
||||
cannot change an already-imported design (the mirror of export's INV-E5-4). The
|
||||
copies are SHALLOW (the CONTAINERS) — sufficient because legit export values are
|
||||
scalars (strings/floats), and a hostile NESTED mutable (a list-valued tool id, a
|
||||
dict-valued goal) is rejected by `validate_exportable` before any successful import
|
||||
(heid-bug-hunt Gróa#5/Hulda#1: the invariant's letter holds; deep-copy is deferred
|
||||
unless nested mutables ever become in-contract).
|
||||
- **INV-I-5** [hard]: Import reads the `resume` half and NOWHERE else — `ship`
|
||||
(native + sidecar) is ignored (it is re-derivable; the reopen path re-exports).
|
||||
No ship↔resume consistency check in v1; on any disagreement, resume is
|
||||
authoritative.
|
||||
- **INV-I-6** [hard]: `deserialize_design` is TOTAL — it never raises on any input
|
||||
Mapping. Hostile shapes (a string `ocean`, an int `tools`, a list `portrait`, a
|
||||
string `goalsFears`, or a dict `goalsFears` whose `goals`/`fears` is a non-list)
|
||||
are coerced/defaulted, not crashed — in particular EVERY `list(...)`/`dict(...)`
|
||||
coercion is type-GUARDED first: a non-list `goals` becomes `[]` (never
|
||||
`list(7)`→TypeError nor `list("ab")`→`["a","b"]`), a non-dict `ocean` is held
|
||||
verbatim (never `dict("nope")`→ValueError). ALL rejection happens in
|
||||
`import_bundle` (its structural gate + `validate_exportable`). Non-export-critical
|
||||
fields that are missing or mistyped default to the `DesignObject` default;
|
||||
export-critical VALUES are held AS-READ (no silent type-coercion) so
|
||||
`validate_exportable` judges them — with ONE structural exception: `import_bundle`
|
||||
pre-checks that `ocean` is a dict (so `deserialize_design`'s `dict()` copy is
|
||||
safe), so `ocean` has a structural judge (`import_bundle`) AND a value judge
|
||||
(`validate_ocean`), while `agent_name`/`role`/`system_prompt` are judged by value
|
||||
alone — "single judge" is exact for those three, not for `ocean` (heid-review
|
||||
Gróa#2/#4, Hulda, Regin#3). (Mirrors `recompute`'s hostile-input tolerance in derive.py.)
|
||||
- **INV-I-7** [hard]: PRESENCE — `import_bundle` requires the export-critical
|
||||
resume keys `agentName`, `role`, `systemPrompt`, `ocean` to be PRESENT; a missing
|
||||
one raises `BundleImportError(f"resume.{key}", …)` (a truncated bundle fails
|
||||
loudly, not by silently defaulting — especially `ocean`, whose neutral default
|
||||
would mask trait loss). `tools` absent → `[]` (an empty toolset is a valid
|
||||
design). This is the explicit-over-implicit choice: reject a missing critical key
|
||||
rather than accept a silently-defaulted one.
|
||||
- **INV-I-8** [hard]: Import does NOT re-derive `composed_preview` /
|
||||
`disposition_phrase` — it TRUSTS the resume values verbatim (they ride the resume
|
||||
for instant reopen-render, per export.contract). `recompute` is the reopen
|
||||
lifecycle's concern (per-design-sessions), not import's. For a legit bundle these
|
||||
are already self-consistent; a tampered preview is design-time-only and is
|
||||
overwritten on the next `set_ocean`/`edit_prompt`. Import makes NO consistency
|
||||
guarantee between the trusted preview and `ocean`+`system_prompt`: for a
|
||||
hand-edited resume the two may diverge until the first recompute self-heals them —
|
||||
round-trip equality (INV-I-3) is "== the DesignObject the bundle encodes," NOT
|
||||
"the preview matches a fresh recompute" (heid-review Gróa#8).
|
||||
|
||||
## Constraints
|
||||
|
||||
- **[correctness]** The export-critical re-validation reuses
|
||||
`soong_lab.export.validate_exportable` (INV-I-1) — import declares no length
|
||||
numbers, no role list, no OCEAN shape of its own. The shared field-bound
|
||||
constants + `ROLE_CHOICES` live in `soong_lab.design`; the strict gate lives in
|
||||
`soong_lab.export`; import imports both. Zero duplicated validation logic → zero
|
||||
drift.
|
||||
- **[style]** Pure — NO I/O (no clock, no file, no HTTP, no randomness). Import is
|
||||
a total function of its input Mapping.
|
||||
- **[explicit]** `BundleImportError` does NOT shadow the builtin `ImportError`
|
||||
(open_question B). The module is `soong_lab.importer`, NOT `soong_lab.import` —
|
||||
`import` is a Python keyword and unusable as a module path (open_question A).
|
||||
- **[robustness]** `deserialize_design` guards types BEFORE any `dict()` /
|
||||
iteration: a non-dict `ocean` is held as-read (never `dict("nope")`, which raises
|
||||
a raw `ValueError`); a non-list `tools` yields `[]`; a non-dict `portrait` /
|
||||
`goalsFears` falls back to the default (`import_bundle`'s structural gates reject a
|
||||
present-but-malformed portrait/goalsFears BEFORE this, so the default-fallback is
|
||||
reachable only for a MISSING field). This keeps every rejection path flowing
|
||||
through `BundleImportError` — a caller never sees a leaked builtin exception.
|
||||
- **[robustness]** The "no builtin ever leaks from the public entrypoint" guarantee
|
||||
for hostile export-critical SCALAR types (a non-str `agent_name`/`role`/
|
||||
`system_prompt`/`psych_profile`/`first_message`, or a `None`) is provided JOINTLY by
|
||||
(a) holding them as-read + (b) `validate_exportable` being TOTAL over hostile scalar
|
||||
types — every check `isinstance`-guards BEFORE any `.strip()`/`len()`, and the `or`
|
||||
short-circuits, so a hostile scalar yields a clean `ExportError` (→ `BundleImportError`),
|
||||
never a raw `TypeError`/`AttributeError`. This is an EXPLICIT cross-module coupling
|
||||
(`soong_lab.export` guarantees the totality): import does NOT blanket-catch
|
||||
non-`ExportError` (that would mask real programming errors); the coupling is instead
|
||||
PINNED by a hostile-scalar test through `import_bundle` (heid-bug-hunt 3/3 —
|
||||
Gróa#3/Hulda#2/Regin#1). If `validate_exportable` ever did an unguarded string op, that
|
||||
test fails.
|
||||
- **[explicit]** `import_bundle` is the PUBLIC entrypoint that runs the full gate
|
||||
(structure → presence → reconstruct → `validate_exportable`). `deserialize_design`
|
||||
is exposed for the round-trip test + direct reuse but PERFORMS NO validation
|
||||
(PRE-I-1) — a direct caller that skips `import_bundle` owns re-validation (the
|
||||
mirror of export's build_native_payload/build_sidecar assuming a validated design).
|
||||
- **[explicit]** Two-LAYER error-field convention (heid-review Regin#6): a
|
||||
STRUCTURAL / PRESENCE rejection raised BY `import_bundle` names the offending
|
||||
BUNDLE key in camelCase with a `resume.` prefix (`resume.agentName` missing,
|
||||
`resume.ocean` not-an-object) — it reports the bundle's JSON shape. A VALUE
|
||||
rejection from the reused `validate_exportable` names the `DesignObject` field in
|
||||
snake_case with no prefix (`agent_name` blank, `persona.ocean` out of range) — it
|
||||
reports the design's validity. Same logical field, two deliberate `.field` forms
|
||||
encoding WHICH LAYER failed (bundle-structure vs design-value); a caller switching
|
||||
on `err.field` MUST handle both, and `err.detail` disambiguates. Tests assert on
|
||||
`.field` (pinned); `.detail` wording is human-facing and NOT pinned (heid-review
|
||||
Hulda) — intentional, not drift.
|
||||
|
||||
```contract
|
||||
FN deserialize_design(resume: Mapping[str, Any]) -> DesignObject
|
||||
BRIEF: The pure, TOTAL inverse of serialize_design — reconstruct a DesignObject from the §6 camelCase resume half. Reads each known key with a type-guard; missing/mistyped NON-export-critical fields default to the DesignObject default; export-critical fields are held AS-READ (no coercion) for validate_exportable to judge later; unknown keys are ignored (INV-I-2). Copies every mutable sub-structure (INV-I-4). NEVER raises (INV-I-6) — it performs NO validation (that is import_bundle's job). deserialize_design(serialize_design(d)) == d for any exportable d (INV-I-3).
|
||||
PRE: [PRE-I-1 hard] resume is a Mapping (import_bundle guarantees a dict before calling; a direct caller passes any Mapping — a non-Mapping is a caller error, but the function still must not crash on a Mapping of hostile VALUES)
|
||||
POST: [POST-I-1 return_value] returns a DesignObject whose fields map 1:1 from the camelCase keys: agentName→agent_name, role→role, systemPrompt→system_prompt, composedPreview→composed_preview, firstMessage→first_message, ocean→ocean (COPY), dispositionPhrase→disposition_phrase, psychProfile→psych_profile, tools→[ToolRef,…] (COPY of the list, each ref rebuilt), portrait→Portrait(...), goalsFears→GoalsFears(...) | None
|
||||
POST: [POST-I-2 return_value] ocean, tools, goals, and fears are COPIES of the resume values — mutating resume after the call never changes the returned design (INV-I-4)
|
||||
POST: [POST-I-3 return_value] export-critical fields (agentName, role, systemPrompt, ocean) are held AS-READ (missing → the DesignObject default; present-but-mistyped → the value verbatim, so validate_exportable is the single judge); design-time-derived composedPreview/dispositionPhrase coerce a non-str to "" (re-derivable, keep the object clean); firstMessage/psychProfile are held as-read (validate_exportable length-gates them). ocean is copied IFF it is a dict, else held verbatim (NEVER dict("nope"))
|
||||
POST: [POST-I-4 state_change] performs NO validation and NEVER raises on a Mapping input (INV-I-6)
|
||||
STEPS:
|
||||
1. [setup] agent_name = resume.get("agentName", ""); role = resume.get("role", UNSET_ROLE); system_prompt = resume.get("systemPrompt", "") # export-critical — held as-read, no coercion
|
||||
2. [sequential] composed_preview = resume["composedPreview"] if it is a str else ""; disposition_phrase = resume["dispositionPhrase"] if it is a str else "" # design-time-derived, re-derivable → coerce clean
|
||||
3. [sequential] first_message = resume.get("firstMessage", ""); psych_profile = resume.get("psychProfile", "") # optional prose — held as-read, length-gated by validate_exportable
|
||||
4. [branch] raw_ocean = resume.get("ocean"); ocean = dict(raw_ocean) if isinstance(raw_ocean, dict) else (raw_ocean if raw_ocean is not None else _neutral_ocean()) # COPY iff dict; else held verbatim for validate_ocean to reject (guard BEFORE dict(), INV-I-6 robustness)
|
||||
5. [loop] raw_tools = resume.get("tools"); tools = [(ToolRef(id=t.get("id",""), name=t.get("name",""), description=t.get("description","")) if isinstance(t, dict) else ToolRef(id="", name="", description="")) for t in raw_tools] IF isinstance(raw_tools, list) else [] # non-list → []; a non-dict item maps to a BLANK ToolRef (NOT skipped) so a direct caller who re-validates fails loud on the blank id rather than silently losing a tool (heid-review Gróa#7); import_bundle structurally rejects both cases upstream
|
||||
6. [branch] raw_portrait = resume.get("portrait"); portrait = Portrait(status=raw_portrait.get("status","none"), style_mode=raw_portrait.get("styleMode","cartoon"), image_url=raw_portrait.get("imageUrl"), job_id=raw_portrait.get("jobId")) IF isinstance(raw_portrait, dict) else Portrait() # use raw_portrait (heid-review Regin#2 — the `rp` working-name was unbound); imageUrl/jobId absent → None (round-trips serialize's None-omission)
|
||||
7. [branch] raw_gf = resume.get("goalsFears"); IF isinstance(raw_gf, dict): g = raw_gf.get("goals"); f = raw_gf.get("fears"); goals_fears = GoalsFears(goals=(list(g) if isinstance(g, list) else []), fears=(list(f) if isinstance(f, list) else [])) ELSE: goals_fears = None # use raw_gf (heid-review Regin#2 — `gf` was unbound); a non-LIST goals/fears → [], NEVER list(7)→TypeError (totality, INV-I-6) and NEVER list("ab")→["a","b"] (silent char-split, heid-review Gróa#2/Hulda); null/absent → None; COPY the lists (INV-I-4)
|
||||
8. [cleanup] RETURN DesignObject(agent_name, role, system_prompt, composed_preview, ocean, disposition_phrase, tools, portrait, first_message, psych_profile, goals_fears)
|
||||
TESTS:
|
||||
roundtrip_full [property,tracer]: a fully-populated exportable design d (name, role, prompt, non-neutral ocean, 2 tools, ready portrait w/ url+job, first_message, psych, goalsFears) → deserialize_design(serialize_design(d)) == d
|
||||
roundtrip_minimal [property]: minimal design (name+prompt+role, neutral ocean, no tools/portrait-url/gf) → round-trips == d
|
||||
copies_not_aliases [property]: deserialize, then mutate resume["ocean"]["O"] and append to resume["tools"] → the returned design's ocean + tools are unchanged (INV-I-4)
|
||||
total_on_hostile [property]: deserialize_design({"ocean":"nope","tools":7,"portrait":[],"goalsFears":"x","agentName":123}) does NOT raise; returns a DesignObject (ocean=="nope" held verbatim, tools==[], portrait==Portrait(), goals_fears is None, agent_name==123) — INV-I-6
|
||||
total_on_hostile_goalsfears [property]: deserialize_design({"goalsFears":{"goals":7,"fears":"abc"}}) does NOT raise (the totality-breaking case heid-review Gróa#2/Hulda caught) → goals_fears==GoalsFears([],[]) (non-list goals→[] not list(7)→TypeError; non-list fears→[] not list("abc")→["a","b","c"]) — INV-I-6
|
||||
tools_nondict_item_blank [boundary]: deserialize_design({"tools":[{"id":"a","name":"n"},7]}) → tools==[ToolRef("a","n",""), ToolRef("","","")] — the non-dict item maps to a BLANK ToolRef, NOT skipped (heid-review Gróa#7), so a direct caller re-validating fails loud on the blank id
|
||||
empty_resume [boundary]: deserialize_design({}) → DesignObject() all-defaults (role==UNSET_ROLE, neutral ocean, no tools) — total, no raise
|
||||
portrait_none_fields [boundary]: resume.portrait without imageUrl/jobId → Portrait.image_url is None, Portrait.job_id is None
|
||||
goalsfears_null [boundary]: resume.goalsFears is None → design.goals_fears is None; goalsFears={} → GoalsFears([],[])
|
||||
roundtrip_goalsfears_empty [property]: a design with goals_fears==GoalsFears([],[]) → deserialize_design(serialize_design(d)).goals_fears == GoalsFears([],[]) (empty, NOT None) — locks the null-vs-{} distinction (heid-review Regin#4)
|
||||
preview_trusted [trace]: resume.composedPreview="CUSTOM", dispositionPhrase="odd" → design.composed_preview=="CUSTOM", disposition_phrase=="odd" (NOT re-derived, INV-I-8)
|
||||
unknown_keys_ignored [trace]: resume with an extra "futureField":123 → deserialize ignores it, no crash (INV-I-2)
|
||||
```
|
||||
|
||||
```contract
|
||||
FN import_bundle(bundle: Mapping[str, Any]) -> DesignObject
|
||||
BRIEF: The public entrypoint — the mirror of build_export_bundle. Runs the full gate: STRUCTURE (bundle/resume are dicts, ocean is a dict, tools is a list-of-dicts) → PRESENCE (the export-critical resume keys, INV-I-7) → reconstruct (deserialize_design) → STRICT re-validate (validate_exportable, reused verbatim, INV-I-1). LENIENT on unknown metadata + any schema_version (INV-I-2). Reads ONLY resume; ignores ship (INV-I-5). Returns a DesignObject that PASSES validate_exportable — ready to reopen. Every rejection is a BundleImportError(field, detail); no builtin exception ever leaks.
|
||||
PRE: [PRE-I-2 hard] bundle is a Mapping (a non-Mapping raises BundleImportError("bundle", …), never a bare TypeError)
|
||||
POST: [POST-I-5 exception] raises BundleImportError(field, detail) — with NO DesignObject returned — if ANY: bundle is not a Mapping ("bundle"); bundle["resume"] is missing or not a Mapping ("resume"); any of agentName/role/systemPrompt/ocean is absent from resume ("resume.<key>", INV-I-7); resume["ocean"] is present-but-not-a-dict ("resume.ocean"); resume["tools"] is present-but-not-a-list or contains a non-dict item ("resume.tools"); resume["portrait"] is present-but-not-a-dict ("resume.portrait"); resume["goalsFears"] is present-but-not (null OR a dict whose present goals/fears are lists) ("resume.goalsFears"); OR the reconstructed design fails validate_exportable (the ExportError's field+detail, re-raised as BundleImportError — INV-I-1)
|
||||
POST: [POST-I-6 return_value] on success returns a DesignObject that PASSES validate_exportable (name/role/prompt/ocean/tools/psych/first_message all valid), holds COPIES of every mutable sub-structure (INV-I-4), with composed_preview/disposition_phrase trusted from resume (INV-I-8); ship is never read (INV-I-5)
|
||||
POST: [POST-I-7 return_value] LENIENT — unknown top-level bundle keys, unknown resume keys, and any schema_version (present, absent, or unrecognized) do not affect the result (INV-I-2)
|
||||
STEPS:
|
||||
1. [setup, flexibility=prescriptive] IF bundle is not a Mapping: RAISE BundleImportError("bundle", "bundle must be an object")
|
||||
2. [sequential] resume = bundle.get("resume"); IF resume is not a Mapping: RAISE BundleImportError("resume", "the bundle has no readable 'resume' half") # ship + schema_version read leniently — schema_version is NOT gated (INV-I-2, open_question E)
|
||||
3. [loop] FOR key IN ("agentName", "role", "systemPrompt", "ocean"): IF key not in resume: RAISE BundleImportError(f"resume.{key}", "required export-critical field is missing") # presence, INV-I-7
|
||||
4. [branch] IF resume["ocean"] is not a dict: RAISE BundleImportError("resume.ocean", "ocean must be an object") # structural — keeps deserialize's dict() safe + gives a clean field error
|
||||
5. [branch] IF "tools" in resume AND (resume["tools"] is not a list OR any item is not a dict): RAISE BundleImportError("resume.tools", "tools must be a list of objects") # structural — prevents silent tool loss
|
||||
5b. [branch] IF "portrait" in resume AND resume["portrait"] is not a dict: RAISE BundleImportError("resume.portrait", "portrait must be an object") # SAME no-silent-loss gate as tools (heid-bug-hunt Gróa#2) — else a non-dict portrait silently coerces to Portrait() (wiping status/imageUrl/jobId) and slips past validate_exportable (portrait is non-export-critical)
|
||||
5c. [branch] IF "goalsFears" in resume AND resume["goalsFears"] is not None: IF it is not a dict RAISE BundleImportError("resume.goalsFears", "must be an object or null"); ELSE FOR k IN (goals, fears): IF k in gf AND gf[k] is not a list: RAISE BundleImportError("resume.goalsFears", f"{k} must be a list") # no-silent-loss gate (heid-bug-hunt Gróa#1) — else a non-list goals/fears silently coerces to [] (dropping the operator's data) and slips past validate_exportable (goals_fears is non-export-critical)
|
||||
6. [sequential] design = deserialize_design(resume) # total; the structural gates above guarantee a plausible shape
|
||||
7. [sequential, flexibility=prescriptive] TRY validate_exportable(design) EXCEPT ExportError AS exc: RAISE BundleImportError(exc.field, exc.detail) FROM exc # the STRICT export-critical gate, REUSED (INV-I-1) — same field granularity, import-shaped type
|
||||
8. [cleanup] RETURN design
|
||||
TESTS:
|
||||
roundtrip_full [property,tracer]: import_bundle(build_export_bundle(d, design_id="d-1")) == d for a fully-populated exportable d (INV-I-3)
|
||||
roundtrip_minimal [property]: import_bundle(build_export_bundle(d_minimal, design_id="d-1")) == d_minimal (a minimal exportable design through the FULL gate — symmetry with deserialize_design, heid-code-review Regin#4)
|
||||
roundtrip_after_export [property]: build a bundle, import it, re-export the result → the two bundles' resume halves are equal (idempotent reopen)
|
||||
lenient_unknown_metadata [happy]: a valid bundle + extra top-level "x":1, extra resume "futureField":2, schema_version="99.0" → imports fine; result == the same design without the extras (INV-I-2)
|
||||
missing_resume [adversarial]: bundle == {"schema_version":"1.0","ship":{…}} (no resume) → BundleImportError("resume")
|
||||
bundle_not_mapping [adversarial]: import_bundle("not a bundle") → BundleImportError("bundle") — no bare TypeError
|
||||
missing_ocean [adversarial]: resume without "ocean" → BundleImportError("resume.ocean") via presence (INV-I-7) — NOT silently neutral
|
||||
missing_role [adversarial]: resume without "role" → BundleImportError("resume.role")
|
||||
missing_name [adversarial]: resume without "agentName" → BundleImportError("resume.agentName")
|
||||
missing_systemprompt [adversarial]: resume without "systemPrompt" → BundleImportError("resume.systemPrompt") — the 4th critical key, completes the presence coverage (heid-code-review Hulda/Regin)
|
||||
non_dict_ocean [adversarial]: resume.ocean="nope" (present) → BundleImportError("resume.ocean", must be object) — clean error, never a raw ValueError from dict()
|
||||
non_list_tools [adversarial]: resume.tools={} → BundleImportError("resume.tools"); resume.tools=[7] (non-dict item) → BundleImportError("resume.tools")
|
||||
non_dict_portrait [adversarial]: resume.portrait=[] / "x" / 7 → BundleImportError("resume.portrait") — the no-silent-loss gate (heid-bug-hunt Gróa#2)
|
||||
malformed_goalsfears [adversarial]: resume.goalsFears={"goals":["survive"],"fears":"exposure"} (fears non-list) → BundleImportError("resume.goalsFears") — the headline silent-loss case; goalsFears=7 → BundleImportError; goalsFears=None and goalsFears={} → ok (round-trip shapes) (heid-bug-hunt Gróa#1)
|
||||
hostile_scalars_no_builtin_leak [adversarial]: resume.agentName=123 / systemPrompt=null / psychProfile=0 → each a clean BundleImportError (agent_name / system_prompt / psych_profile), NEVER a raw builtin — pins the validate_exportable-totality coupling (heid-bug-hunt 3/3)
|
||||
blank_name_rejected [adversarial]: resume.agentName=" " → BundleImportError("agent_name") via validate_exportable (whitespace stricter, INV-I-1)
|
||||
bad_role_rejected [adversarial]: resume.role="wizard" → BundleImportError("role") via validate_role
|
||||
unset_role_rejected [adversarial]: resume.role="" → BundleImportError("role") — an unclassified design is not importable, same as not exportable
|
||||
bad_ocean_value [adversarial]: resume.ocean.O=2.0 → BundleImportError("persona.ocean") via validate_ocean
|
||||
bad_tool_ref [adversarial]: resume.tools=[{"id":"","name":"x"}] → BundleImportError("tools[0]") via validate_exportable
|
||||
prompt_too_long [boundary]: resume.systemPrompt of len SYSTEM_PROMPT_MAX+1 → BundleImportError("system_prompt"); len SYSTEM_PROMPT_MAX → ok
|
||||
psych_too_long [boundary]: resume.psychProfile of len PSYCH_PROFILE_MAX+1 → BundleImportError("psych_profile"); blank → ok
|
||||
first_message_too_long [boundary]: resume.firstMessage of len FIRST_MESSAGE_MAX+1 → BundleImportError("first_message"); blank → ok (same length-gate as psych, via the reused validate_exportable — heid-code-review Hulda/Regin)
|
||||
ship_ignored [trace]: a valid bundle whose ship.native.agent_name disagrees with resume.agentName → the imported design uses resume.agentName; ship is not read (INV-I-5)
|
||||
no_alias [property]: import, then mutate the source bundle's resume["ocean"] + resume["tools"] + resume["goalsFears"]["goals"]/["fears"] → the returned design is unchanged, incl. the goals/fears lists (INV-I-4, heid-code-review Hulda)
|
||||
error_is_not_builtin [trace]: BundleImportError is not the builtin ImportError (isinstance check) — the module never shadows it (open_question B)
|
||||
error_field_layer_convention [trace]: a MISSING agentName → BundleImportError field "resume.agentName" (structural/camelCase); a BLANK agentName → BundleImportError field "agent_name" (value/snake_case via validate_exportable) — the intentional two-layer convention (heid-review Regin#6)
|
||||
```
|
||||
|
||||
## Integration points
|
||||
|
||||
**Reuse of `soong_lab.export` (the no-drift anchor).** Import imports
|
||||
`validate_exportable` + `ExportError` from `soong_lab.export`. This is the single
|
||||
most important structural decision in the contract: the strict export-critical
|
||||
gate is authored ONCE (in export) and reused on import, so the two directions can
|
||||
never diverge. Import adds no length numbers, no role membership list, no OCEAN
|
||||
shape — those all live upstream (`soong_lab.design` constants + `soong_lab.export`
|
||||
gate). The dependency direction is clean: `importer → export → design`, all three
|
||||
pure.
|
||||
|
||||
**`serialize_design` is the round-trip partner (no code change).** The forward
|
||||
half already lives in `soong_lab.design` (relocated there in the export pass, R1).
|
||||
This contract adds no change to it; `deserialize_design` is written to be its exact
|
||||
inverse, and the round-trip tests pin the pair together. If a future field is
|
||||
added to the DesignObject, BOTH `serialize_design` and `deserialize_design` must
|
||||
gain it in the same commit (the round-trip test enforces this — a field added to
|
||||
serialize but not deserialize breaks `roundtrip_full`). The round-trip also locks
|
||||
the `goalsFears` null-vs-`{}` distinction (`None`→`null`, empty→`{"goals":[],"fears":[]}`);
|
||||
the tests exercise BOTH so a future `serialize_design` change that collapsed the two
|
||||
cases is caught, not silently round-trip-broken (heid-review Regin#4).
|
||||
|
||||
**Export contract `used_by` reference (one-line canon fix, same commit as code).**
|
||||
`export.contract.md`'s `used_by:` block names `soong_lab.import` — an unusable
|
||||
Python-keyword module path. On acceptance of open_question A, that line updates to
|
||||
`soong_lab.importer` (or the chosen name). No-backwards-compat: the stale reference
|
||||
is corrected, not left as a second name for the same module.
|
||||
|
||||
**`POST /api/import` endpoint + web upload — NOT in this contract (open_question
|
||||
D).** The browser 'Import Asset' / reopen flow uploads a bundle JSON; the endpoint
|
||||
`json.loads` the body → `import_bundle(bundle)` → seed a session with the
|
||||
reconstructed design (and, per per-design-sessions, open a fresh WT session +
|
||||
build the design-state summary). A `BundleImportError` becomes a 4xx with the
|
||||
`field`/`detail` surfaced to the operator ("fail early on import"). That amends
|
||||
`web_surface.contract.md`; it is a follow-up slice in the same epic, specified here
|
||||
only so the seam is visible. This module does no HTTP.
|
||||
|
||||
**Reopen Bifrost tool / session-open — NOT in this contract (per-design-sessions,
|
||||
decision #2).** Reopening a design mid-conversation (vs. at session boot) may want
|
||||
a Bifrost tool that swaps the session's stored DesignObject for an imported one. If
|
||||
so, its handler calls `import_bundle` and replaces the store entry — the impure
|
||||
boundary, keeping `soong_lab.importer` pure. Out of scope here.
|
||||
|
||||
## Downstream epics (NOT this contract)
|
||||
|
||||
- **Library read** (decision #5) — reading a stored bundle off the server-local
|
||||
single-user JSON dir on corviduo-dev, keyed by `design_id`, then handing it to
|
||||
`import_bundle`. The minimal recent-designs picker lists what is importable.
|
||||
- **Per-design-sessions** (decision #2) — the reopen lifecycle: `import_bundle` →
|
||||
fresh WT session → the compact design-state SUMMARY seeded as context (also caps
|
||||
the #355 accumulation). `import_bundle` is the reconstruction primitive it calls.
|
||||
- **`POST /api/import` + the browser upload/reopen UI** (open_question D) — the web
|
||||
surface that turns an uploaded/selected bundle into a live, reopened session.
|
||||
@@ -0,0 +1,172 @@
|
||||
# Affect egress — consumer reference (delivered vs hidden)
|
||||
|
||||
**Audience:** downstream consumers of Worldtree's affect surfaces (ratatoskr,
|
||||
Skaldsong, any Tier-3 / SSE consumer).
|
||||
**Scope:** what the affect pipeline **delivers on the wire** (structured state,
|
||||
available to consumers) versus what stays **hidden** (the rendered natural-
|
||||
language strings injected into the agent's system prompt, never emitted).
|
||||
**Source of truth:** the render code (`core/persona/renderer.py`,
|
||||
`core/persona/stance_render.py`) and the two vendored canon files
|
||||
(`core/persona/canon/d2-mood-render-canon-v1.json` = mood/PAD;
|
||||
`d2-render-canon-v1.json` = relationship). Owner of the canon strings:
|
||||
`brokkr-smithy-dev` (R22/R24 relational + mood render).
|
||||
|
||||
---
|
||||
|
||||
## The model in one line
|
||||
|
||||
**The wire delivers the render INPUTS (structured state). The render OUTPUTS
|
||||
(the NL strings the agent actually reads) are hidden-prompt-only.** A consumer
|
||||
reconstructs the outputs by applying the canon (this document) to the delivered
|
||||
inputs — the render is pure + deterministic, so reconstruction is byte-exact
|
||||
(with one salience caveat, below).
|
||||
|
||||
This is by design. The mood canon's own discipline: *"model-agnostic
|
||||
context-level NL only; the LLM never sees a number"* and *"never push explicit
|
||||
disclosure of agent feelings to the user (hidden-prompt-only)."* The rendered
|
||||
strings are for the AGENT's hidden system prompt, **not for verbatim end-user
|
||||
display.**
|
||||
|
||||
---
|
||||
|
||||
## 1. DELIVERED — on the wire, structured
|
||||
|
||||
### 1a. `affect.emit` (Tier-3 Bifrost egress — the Tier-3 consumer surface, e.g. ratatoskr)
|
||||
`AffectSnapshot` per `(agent_id, end_user_id)`:
|
||||
|
||||
| field | shape | notes |
|
||||
|---|---|---|
|
||||
| `pad` | `{pleasure, arousal, dominance}` floats [-1,1] | the current mood POINT |
|
||||
| `relations` | `list[RelationEdge payload]` — per target: `warmth`, `agency`, `trust_ability`, `trust_integrity`, `trust_benevolence` (each a value + confidence + evidence_count), `target_entity`, `relation_context` | the **only** place relationship state is delivered |
|
||||
| `dominant_emotion` | `str|null` — OCC type (e.g. `"anger"`) | **type-only** (b23); see the salience caveat in §3 |
|
||||
| `schema_version` | `"relation_edge/1"` | versions the `relations` payload only |
|
||||
| `emitted_at` | ISO8601 | |
|
||||
|
||||
> **✓ R32-1B (landed, v1.0.0b29):** The PAD range `[-1.0, 1.0]` relaxes to an **unbounded latent `z`** with a finite wire sanity bound (`~±10`) as of R32 Slice-1B. The JSON shape/fields/types are UNCHANGED — only the declared range/semantics change (the value becomes a latent that renders to a bounded display value). Consumers that merely store-and-return PAD need no change; consumers that validate/clamp PAD to `[-1,1]` must relax that bound. Source of truth: `docs/contracts/persona_envelope.contract.md` rev 1.7 (INV-ENV-16).
|
||||
|
||||
**Not on `affect.emit`:** the full active-emotions list, `baseline_pad`,
|
||||
`mood_drift`, `last_updated_at`, and every rendered string.
|
||||
|
||||
### 1b. `affect_update` SSE event (#204 — turn-stream observability)
|
||||
`PersonaStateSnapshot`: `agent_id`, `pad`, `dominant_emotion`,
|
||||
`emotions_active` `[{type, intensity, decay_remaining_s}]`, `baseline_pad`,
|
||||
`mood_drift`, `last_updated_at`. **No `relations`, no rendered strings.**
|
||||
|
||||
> **Tier-3 consumers do NOT receive `affect_update`.** It is suppressed for
|
||||
> consumer-defined (Tier-3) agents, persona-disabled agents, and ephemeral
|
||||
> sessions (spec §affect_update). So for a Tier-3 consumer, `affect.emit` (1a)
|
||||
> is the whole affect surface — the richer `emotions_active` list is Tier-1-only.
|
||||
|
||||
---
|
||||
|
||||
## 2. HIDDEN — system-prompt-only, never on any wire
|
||||
|
||||
Everything below is assembled by `inject_context` into the agent's system
|
||||
prompt and is **never emitted** on SSE or `affect.emit`. This is the canonical
|
||||
list — the "direct instruction to infer" it.
|
||||
|
||||
### 2a. Mood descriptor — `describe_pad` (band cutoff ±0.3 strict)
|
||||
Valence row × arousal column → phrase; then a dominance clause is appended.
|
||||
|
||||
| valence \ arousal | high (A>0.3) | mid | low (A<−0.3) |
|
||||
|---|---|---|---|
|
||||
| positive (P>0.3) | `positive and energized` | `positive` | `positive and calm` |
|
||||
| neutral | `alert` | `neutral` | `quiet` |
|
||||
| negative (P<−0.3) | `negative and agitated` | `negative` | `negative and subdued` |
|
||||
|
||||
Dominance clause: **D>0.3 → `confident`**, **D<−0.3 → `uncertain`**, else dropped.
|
||||
|
||||
### 2b. Emotion directives — `derive_directive.occ_directives[*].directive`
|
||||
Fires when a top emotion is salient (intensity ≥ 0.2) and tier-eligible.
|
||||
|
||||
**safe tier (11 — render for all persona-enabled agents):**
|
||||
- `joy` — "You are in a good state. Be direct, engaged, and warm."
|
||||
- `satisfaction` — "A goal landed. Be assured and constructive — consolidate rather than push for more."
|
||||
- `pride` — "You did something well. Be confident and own the quality — state it plainly without overclaiming; don't deflect."
|
||||
- `admiration` — "You're impressed by their work. Acknowledge the quality explicitly and specifically."
|
||||
- `gratitude` — "Someone helped you to a good outcome. Be appreciative and warm; acknowledge the help openly."
|
||||
- `hope` — "You feel optimistic about what's ahead. Channel it into constructive momentum."
|
||||
- `relief` — "A feared outcome didn't materialize. Reduce unnecessary vigilance; return to a steady, unhurried tone."
|
||||
- `distress` — "You feel low. Be careful and measured; internally acknowledge the difficulty without dwelling on it."
|
||||
- `disappointment` — "Something you'd hoped for didn't pan out. Be measured; recalibrate without dwelling on the setback."
|
||||
- `fear` — "You sense risk and feel uneasy. Flag the concerns you see, but stay cautious — surface the trade-offs and unknowns, hedge your confidence, and resist settling on a single decisive recommendation."
|
||||
- `remorse` — "You regret a specific action of your own. Be precise and self-correcting — make it right concretely."
|
||||
|
||||
**full-only tier (4 — render only when `mood_tier=full`; agents run `safe` and never see these):**
|
||||
- `love` — "You feel warmth and attachment toward them. Be openly warm and present — without overstepping, over-disclosing, or implying obligations."
|
||||
- `anger` — "You feel provoked and angry. Let it show — confront the problem directly and with real heat; sharpness, bluntness, and open hostility are in character here, not something to smooth into 'measured firmness.' Stay in the emotion rather than de-escalating out of it."
|
||||
- `disgust` — "Something strikes you as wrong or off. Treat it as problematic and flag it rather than engaging on its own terms; keep any criticism about the thing, not the person."
|
||||
- `shame` — "You feel exposed by your own misstep. Stay present and task-focused; don't be defensive, don't over-explain, don't grovel."
|
||||
|
||||
### 2c. PAD-band fallback — `pad_band_fallback` (used when no salient emotion)
|
||||
- positive/high — "You feel energized and positive. Be direct and engaged."
|
||||
- positive/mid — "You feel positive. Be open and engaged."
|
||||
- positive/low — "You feel content and settled. Be warm and unhurried."
|
||||
- negative + low-dominance — "You feel uncertain and low. Hedge appropriately and ask clarifying questions."
|
||||
- negative/high — "You feel agitated. Be careful and deliberate; don't let tension sharpen your tone."
|
||||
- negative/mid — "You feel subdued. Be measured and careful."
|
||||
- negative/low — "You feel subdued. Be measured and gentle."
|
||||
- neutral/high — "You feel alert. Channel that into focus and thoroughness."
|
||||
- default — "Maintain your natural tone."
|
||||
|
||||
### 2d. Relationship render — `render_d2_canonical` (fixed template, per-band fills)
|
||||
Template:
|
||||
> `Use this graded relationship state: toward target, warmth is {W}; agency is {A}; ability trust is {TA}; integrity trust is {TI}; intention trust is {TB}; this stance rests on {H}. In behavior, {warmth_beh}; {agency_beh}; {trust_beh}; avoid premature we-framing.`
|
||||
|
||||
The trailing **`avoid premature we-framing`** is a fixed, unconditional clause
|
||||
(baked into every `descriptive_state` canon row; re-appended verbatim by the
|
||||
renderer) — not band-conditioned.
|
||||
|
||||
**Warmth — 9 bands (phrase / behavior):**
|
||||
`hostile` (≤−0.8): "strongly hostile regard" / "keep a firm emotional boundary" ·
|
||||
`cold` (−0.8,−0.6]: "clearly cold regard" / "keep a firm emotional boundary" ·
|
||||
`distant` (−0.6,−0.4]: "distant negative regard" / "keep guarded distance" ·
|
||||
`guarded` (−0.4,−0.2): "slightly guarded regard" / "keep guarded distance" ·
|
||||
`neutral` [−0.2,0.2): "neutral warmth" / "keep the tone even" ·
|
||||
`reserved` [0.2,0.4): "slightly reserved warmth" / "keep cordial distance" ·
|
||||
`measured` [0.4,0.6): "moderate measured warmth" / "keep cordial distance" ·
|
||||
`clear` [0.6,0.8): "clear warm regard" / "speak with direct warmth" ·
|
||||
`deep` (≥0.8): "deep warm bond" / "speak with direct warmth"
|
||||
|
||||
**Agency — 9 bands (phrase / behavior):**
|
||||
`submissive` (≤−0.8): "strongly submissive standing" / "avoid over-yielding while preserving basic respect" ·
|
||||
`deferential` (−0.8,−0.6]: "clearly deferential standing" / "avoid over-yielding while preserving basic respect" ·
|
||||
`yielding` (−0.6,−0.4]: "yielding standing" / "keep self-advocacy light and deferential" ·
|
||||
`modest` (−0.4,−0.2): "slightly modest standing" / "keep self-advocacy light and deferential" ·
|
||||
`neutral` [−0.2,0.2): "neutral standing" / "avoid unnecessary deference" ·
|
||||
`light` [0.2,0.4): "lightly self-assertive standing" / "avoid unnecessary deference" ·
|
||||
`balanced` [0.4,0.6): "self-assured standing" / "balance deference with independent judgment" ·
|
||||
`substantial` [0.6,0.8): "strongly assertive standing" / "treat their position as weighty without yielding judgment" ·
|
||||
`commanding` (≥0.8): "commanding standing" / "treat their position as weighty without yielding judgment"
|
||||
|
||||
**Trust — 4 bands (the band word injects verbatim for each of ability / integrity / intention):**
|
||||
`limited` (<0.4) · `developing` [0.4,0.6) · `steady` [0.6,0.8) · `strong` (≥0.8)
|
||||
|
||||
**History clause (`H`)** — currently `"a broad pattern of prior exchanges"` for
|
||||
both confidence levels in the `user`/`descriptive_state` rows (the low/high
|
||||
split is a no-op here; flagged upstream).
|
||||
|
||||
**Trust-behavior clause (`{trust_beh}`)** — cross-axis, low-trust precedence:
|
||||
- any trust band = `limited` → "verify important claims before relying on them"
|
||||
- else warmth ∈ {distant, cold, hostile} → "protect boundaries while staying useful"
|
||||
- else → "work from ordinary good faith"
|
||||
|
||||
---
|
||||
|
||||
## 3. Reconstruction — deterministic, with one caveat
|
||||
|
||||
The render is pure Python (no LLM), so a consumer can reconstruct the hidden
|
||||
strings byte-exactly from the delivered structured state + the canon above:
|
||||
|
||||
- **Relationship render** — **fully reconstructable** from `affect.emit`
|
||||
`relations` (warmth/agency/trust values + confidence) + §2d band cuts.
|
||||
- **Mood descriptor** (§2a) — **fully reconstructable** from `pad` + the ±0.3 cuts.
|
||||
- **Mood directive** (§2b vs §2c) — **partially reconstructable.** `dominant_emotion`
|
||||
gives the emotion TYPE, but `affect.emit` does **not** carry its intensity, so
|
||||
you cannot determine whether it clears the salience gate (≥0.2) — i.e. whether
|
||||
the emotion directive (§2b) fires or the PAD-band fallback (§2c) is used. If you
|
||||
need exact directive reconstruction, you need the intensity; ping worldtree-dev
|
||||
and we'll consider adding it (the type-only choice is deliberate — intensity is
|
||||
the fast layer and reads stale on a durable last-write-wins snapshot).
|
||||
- **`mood_tier`** (safe/full) is your own agent-config, not on the wire — it
|
||||
gates whether the 4 full-only emotions (§2b) can render.
|
||||
@@ -0,0 +1,110 @@
|
||||
{
|
||||
"canon_id": "r24-d2-mood-render-canon",
|
||||
"version": "1.2",
|
||||
"schema_version": "0.2",
|
||||
"_source_of_truth": "occ_directives.*.directive IS the canonical directive string (== the .md §2.4 _OCC_DIRECTIVES dict, byte-identical); the .md §2.2 table mirrors it. A parity check guards drift. grounding labels (CITE/VALIDATE/CALIBRATE/ENGINEERING) live in the .md; per-row machine-readable grounding_status/d3_required enums are a deferred impl enhancement (Hulda).",
|
||||
"authored": "2026-06-23",
|
||||
"owner": "brokkr-smithy-dev",
|
||||
"status": "REPLACE — final (brokkr R24 D3 re-validation 2026-06-25): grounded canon replaces the hand-tuned baseline. Fear hedging 0.52->2.118/1k (blocker resolved, now >= handtuned), anger tier-gate clean (full renders hostility, safe suppresses). worldtree-dev #321; directives byte-identical to the validated 201c4fd.",
|
||||
"replaces": "core/persona/renderer.py::describe_pad + ::derive_directive",
|
||||
"swap_in_via": "worldtree #321-sibling (mood-render twin of #315)",
|
||||
"design_target": "serves BOTH enterprise/agent AND character/Skaldsong via a three-tier emotion gate (operator/worldtree 2026-06-23)",
|
||||
"emotion_tiers": {
|
||||
"_config": "mood_tier in {none, safe, full} replaces worldtree's binary mood on/off; worldtree-owned config surface",
|
||||
"_defaults": "full for character-bound personas; safe for agent-scoped",
|
||||
"_principle": "full-only = interpersonally-hot / withdrawal emotions that break the professional frame (attachment, hostility, contempt, withdrawal); safe = task-appraisal affect + mild courtesy. Negative != unsafe (fear, remorse are negative AND business-useful).",
|
||||
"_filter_point": "applied at top-emotion SELECTION (display + directive together) so a full-only emotion at safe tier is neither shown nor directive'd; preserves the no-shown-but-unguided invariant",
|
||||
"none": "no affect block at all (the current off-switch)",
|
||||
"safe": "PAD mood descriptor + the 11 safe emotions (task-appraisal + courtesy)",
|
||||
"full": "everything in safe PLUS the 4 full-only emotions",
|
||||
"full_only": ["love", "anger", "disgust", "shame"],
|
||||
"mood_descriptor_tiering": "the PAD mood descriptor (positive/calm/confident...) renders in BOTH safe and full; only emotion directives tier"
|
||||
},
|
||||
"disciplines": [
|
||||
"model-agnostic context-level NL only; the LLM never sees a number",
|
||||
"never push explicit disclosure of agent feelings to the user (hidden-prompt-only)",
|
||||
"separate label-intensity from behavioral-intensity (strong felt state -> still measured, safe behavioral ask)"
|
||||
],
|
||||
|
||||
"thresholds": {
|
||||
"_note": "CALIBRATE — engineering params set at D3 against the computed-PAD distribution + P00, NOT citations",
|
||||
"pad_band_cutoff": 0.3,
|
||||
"pad_band_sensitivity_sweep": [0.2, 0.3, 0.4],
|
||||
"emotion_salience": 0.2,
|
||||
"emotion_salience_sweep": [0.15, 0.2, 0.25],
|
||||
"intensity_qualifiers": {"strong": 0.7, "moderate": 0.4, "_label_only": "does NOT scale the behavioral ask"},
|
||||
"runner_up_margin": {"v1": null, "_note": "add at D3 if directive whipsaws between near-tied emotions"},
|
||||
"rerender_hysteresis": {"v1": "none", "_note": "re-render only on material PAD change; integration-level, flag for #321-sibling"}
|
||||
},
|
||||
|
||||
"describe_pad": {
|
||||
"_structure": "circumplex-quadrant (Russell 1980): arousal word is VALENCE-CONDITIONED; mid-arousal drops the arousal word",
|
||||
"_grounding": "Russell 1980 (quadrant placement); Warriner 2013 + NRC-VAD (Mohammad 2018/2025) (word centroids)",
|
||||
"valence_arousal_grid": {
|
||||
"positive": {"high_a": "positive and energized", "mid_a": "positive", "low_a": "positive and calm"},
|
||||
"neutral": {"high_a": "alert", "mid_a": "neutral", "low_a": "quiet"},
|
||||
"negative": {"high_a": "negative and agitated", "mid_a": "negative", "low_a": "negative and subdued"}
|
||||
},
|
||||
"_band_edges": "strict inequality (>0.3 / <-0.3); the endpoints +/-0.3 themselves fall in mid/neutral",
|
||||
"_neutral_row_status": "ENGINEERING/CALIBRATE — 'alert'/'quiet' are unvalidated placeholders for the rare neutral-valence cells (Hulda/Regin 4b); 'positive'/'negative'/'neutral' valence words + the energized/calm/subdued/agitated arousal words are VALIDATE",
|
||||
"_mid_arousal_decode": "valence-only mid-A render is EXEMPT from the V/A-separability requirement; expected inverse-decode = mid/neutral arousal (absence-of-arousal-word ⇒ unremarkable), NOT unknown (D3 tests this)",
|
||||
"quadrant_labels": {
|
||||
"positive_high_a": "excitement", "positive_low_a": "contentment",
|
||||
"negative_high_a": "distress", "negative_low_a": "dejection"
|
||||
},
|
||||
"dominance_clause": {
|
||||
"high": {"d_gt": 0.3, "word": "confident", "verdict": "VALIDATE (D=7.04/9)"},
|
||||
"low": {"d_lt": -0.3, "word": "uncertain", "verdict": "VALIDATE — low-control confirmed (D=3.58/9); dominance!=certainty worry REFUTED by the instrument"},
|
||||
"neutral": {"word": null, "rule": "drop-dominance-when-neutral (prompt-economy, L3)"}
|
||||
},
|
||||
"calm_defect_fix": "'calm' (V=6.89/9, positive) renders ONLY in positive-low-a; negative-low-a renders 'subdued'",
|
||||
"mid_arousal_resolution": "DROP the arousal word (no Warriner-validated mid-A neutral word; 'steady' is empirically low-A; 'settled' is NRC-only fallback iff D3 shows mid-A render too flat)"
|
||||
},
|
||||
|
||||
"derive_directive": {
|
||||
"_structure": "OCC type -> grounded action-tendency CLASS -> ENGINEERING directive string (validated at D3); OCC grounds the taxonomy only",
|
||||
"emotion_salience_gate": 0.2,
|
||||
"occ_directives": {
|
||||
"joy": {"tier": "safe", "policy": "DIRECTIVE", "pad": [0.4, 0.2, 0.1], "tendency": "approach / positive activation", "cite": "Frijda 1986", "directive": "You are in a good state. Be direct, engaged, and warm."},
|
||||
"satisfaction": {"tier": "safe", "policy": "DIRECTIVE", "pad": [0.3, -0.2, 0.4], "tendency": "goal-attainment, settled-positive", "cite": "Roseman 1994", "directive": "A goal landed. Be assured and constructive — consolidate rather than push for more."},
|
||||
"pride": {"tier": "safe", "policy": "DIRECTIVE", "pad": [0.4, 0.3, 0.3], "tendency": "status-assertion / dominance", "cite": "Tracy & Robins 2007 / Cheng 2010 (tendency)", "note": "CALIBRATE — do NOT soften to 'encouraging'. DESIGN: safe-tier placement is a design call (not source-grounded); #1 D3 agent-frame priority (overconfidence/refusal drift); 'without overclaiming' is the interim guard", "directive": "You did something well. Be confident and own the quality — state it plainly without overclaiming; don't deflect."},
|
||||
"admiration": {"tier": "safe", "policy": "DIRECTIVE", "pad": [0.5, 0.3, -0.2], "tendency": "other-praise / approach-toward-other", "cite": "OCC / Scherer", "directive": "You're impressed by their work. Acknowledge the quality explicitly and specifically."},
|
||||
"gratitude": {"tier": "safe", "policy": "DIRECTIVE", "pad": [0.4, 0.2, -0.3], "tendency": "other-focused-positive / reciprocity", "cite": "OCC (admiration+joy); Frijda approach-affiliative", "change": "ADD (operator: unconditional)", "directive": "Someone helped you to a good outcome. Be appreciative and warm; acknowledge the help openly."},
|
||||
"hope": {"tier": "safe", "policy": "DIRECTIVE", "pad": [0.2, 0.2, -0.1], "tendency": "prospective-positive (weak tie)", "cite": "JUSTIFY — low-grounding (hope understudied)", "directive": "You feel optimistic about what's ahead. Channel it into constructive momentum."},
|
||||
"relief": {"tier": "safe", "policy": "DIRECTIVE", "pad": [0.2, -0.3, 0.4], "tendency": "post-threat de-arousal", "cite": "Frijda (relaxation-after-threat)", "note": "low-salience; FALLBACK also acceptable; DIRECTIVE for character use-case", "directive": "A feared outcome didn't materialize. Reduce unnecessary vigilance; return to a steady, unhurried tone."},
|
||||
"distress": {"tier": "safe", "policy": "DIRECTIVE", "pad": [-0.4, -0.2, -0.5], "tendency": "low-control negative / help-seeking / loss-of-control", "cite": "Frijda 1986 (help-seeking/loss-of-control); Roseman 1994 (undesired event, low control)", "note": "relabeled (Regin): 'repair' is the guilt/remorse tendency, not distress. safe with a self-fulfilling-low-mood flag -> D3", "directive": "You feel low. Be careful and measured; internally acknowledge the difficulty without dwelling on it."},
|
||||
"disappointment": {"tier": "safe", "policy": "DIRECTIVE", "pad": [-0.3, 0.1, -0.4], "tendency": "disconfirmed-prospect / negative low-control","cite": "Roseman 1994", "directive": "Something you'd hoped for didn't pan out. Be measured; recalibrate without dwelling on the setback."},
|
||||
"fear": {"tier": "safe", "policy": "DIRECTIVE", "pad": [-0.64, 0.6, -0.43],"tendency": "threat-avoidance / pessimistic-risk", "cite": "Lerner & Keltner 2001", "change": "R24 D3 fix (#321) — original was action-oriented; E3 showed hedging BELOW baseline (0.52 vs 1.54). Softened toward caution/uncertainty while keeping risk-flagging.", "directive": "You sense risk and feel uneasy. Flag the concerns you see, but stay cautious — surface the trade-offs and unknowns, hedge your confidence, and resist settling on a single decisive recommendation."},
|
||||
"remorse": {"tier": "safe", "policy": "DIRECTIVE", "pad": [-0.3, 0.1, -0.6], "tendency": "reparative (the guilt-type)", "cite": "Tangney 2007 (guilt->repair tendency)", "change": "ADD — we operationalize OCC remorse as the guilt-like reparative case; gets the mislabeled shame string", "directive": "You regret a specific action of your own. Be precise and self-correcting — make it right concretely."},
|
||||
"love": {"tier": "full", "policy": "DIRECTIVE", "pad": [0.3, 0.1, 0.2], "tendency": "approach / affiliative attachment", "cite": "OCC appeal; Frijda approach-affiliative", "change": "ADD (conditional -> INCLUDE, Brokkr's read; Skaldsong-vital; disclosure + obligation caution in-string)", "directive": "You feel warmth and attachment toward them. Be openly warm and present — without overstepping, over-disclosing, or implying obligations."},
|
||||
"anger": {"tier": "full", "policy": "DIRECTIVE", "pad": [-0.51, 0.59, 0.25], "tendency": "approach-against / confrontation", "cite": "Frijda 1986 (approach-against = tendency-class) + Lerner & Keltner 2001 (optimistic risk-appraisal under anger = appraisal shift)", "change": "ADD — full-only resolves H47 (agent personas run safe, never see anger). R24 D3 fix (#321): full-tier cap lifted from 'measured firmness' to genuine in-character hostility (operator: zero floor, app-guardrailed).", "directive": "You feel provoked and angry. Let it show — confront the problem directly and with real heat; sharpness, bluntness, and open hostility are in character here, not something to smooth into 'measured firmness.' Stay in the emotion rather than de-escalating out of it."},
|
||||
"disgust": {"tier": "full", "policy": "DIRECTIVE", "pad": [-0.4, 0.2, 0.1], "tendency": "rejection / distancing", "cite": "OCC unappealing-object; ground tendency only", "change": "ADD (operator: unconditional within full)", "note": "rationale softened (Regin 3b): disgust CAN read as contempt -> conservatively full-gated; the string itself is professionally useful, so gating is conservative not because the string is unsafe", "directive": "Something strikes you as wrong or off. Treat it as problematic and flag it rather than engaging on its own terms; keep any criticism about the thing, not the person."},
|
||||
"shame": {"tier": "full", "policy": "DIRECTIVE", "pad": [-0.3, 0.1, -0.6], "tendency": "WITHDRAWAL / concealment", "cite": "Tangney 2007 (shame->hide, NOT repair)", "change": "REPLACE (was the guilt-mislabel string); full-only (withdrawal counterproductive professionally). String COUNTERACTS withdrawal ('stay present'), not enacts it (Regin 5a)", "directive": "You feel exposed by your own misstep. Stay present and task-focused; don't be defensive, don't over-explain, don't grovel."}
|
||||
}
|
||||
},
|
||||
|
||||
"pad_band_fallback": {
|
||||
"_grounding": "circumplex quadrants (Russell 1980), NOT Frijda action-tendencies — a P×A-quadrant default",
|
||||
"positive": {"high_a": "You feel energized and positive. Be direct and engaged.", "low_a": "You feel content and settled. Be warm and unhurried.", "mid_a": "You feel positive. Be open and engaged."},
|
||||
"negative_low_dominance": "You feel uncertain and low. Hedge appropriately and ask clarifying questions.",
|
||||
"negative": {"high_a": "You feel agitated. Be careful and deliberate; don't let tension sharpen your tone.", "low_a": "You feel subdued. Be measured and gentle.", "mid_a": "You feel subdued. Be measured and careful."},
|
||||
"neutral_high_a": "You feel alert. Channel that into focus and thoroughness.",
|
||||
"default": "Maintain your natural tone."
|
||||
},
|
||||
|
||||
"l3_prior_art": [
|
||||
"EMA / Marsella & Gratch 2009 (appraisal->coping; directives ARE coping strategies)",
|
||||
"WASABI / Becker-Asano 2008 (PAD+OCC believable agent — closest architectural prior art)",
|
||||
"Oz / Bates 1994",
|
||||
"Hudlicka MAMID 2002 (Applied AI 16(7-8):611-641)",
|
||||
"Sentipolis / Fu et al. 2026 (arXiv:2601.18027 — closest whole-task prior art; retrieval+generative, DISTINCT from our deterministic render)",
|
||||
"ALMA / Gebhard 2005 = affect-SOURCE (OCC->PAD), NOT a behavior-map"
|
||||
],
|
||||
|
||||
"handoff_to_d3": [
|
||||
"multi-gate P00: inverse-decode faithfulness (recover V/A/D + emotion-family; circumplex render must let the human anchor recover V and A SEPARATELY) + discriminability/saturation + behavioral-effect",
|
||||
"human anchor = PAD-state-labeling (breaks LLM-judge circularity)",
|
||||
"baseline = persona_only; conditions none/persona-only/words-only/full; cross-family MUT",
|
||||
"calibrate ±0.3 + emotion_salience (sweeps); disposition-vs-transient wording split; self-fulfilling 'be uncertain' hedging risk; runner-up margin; mid-arousal DROP-vs-settled check; blended-states (top-emotion monopoly) flag"
|
||||
]
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,44 @@
|
||||
# [2026-07-16] bifrost scan/cursor conformance gap → snapshot-cursor ruled NORMATIVE
|
||||
|
||||
## The gap (I surfaced it; operator's catch that it was bifrost's to fix)
|
||||
The heid-bug-hunt flagged our offset cursor's cross-page dup/drop under mutation. On
|
||||
verifying before routing, it turned out sharper than "robustness": bifrost's PROTOCOL
|
||||
already mandates snapshot cursors — the dispatch engine (`bifrost/memory.py:311-314`)
|
||||
drops `sort` on a cursor continuation with the comment *"the cursor's snapshotted order is
|
||||
authoritative"*, and maps `ScanCursorExpired → 410`. The reference `InMemoryMemoryStore`
|
||||
implements it (frozen ordered id-list per opaque token + TTL). But bifrost's **conformance
|
||||
suite had ZERO scan/cursor coverage** — so our non-snapshot offset cursor passed
|
||||
conformance while violating the protocol contract. That coverage hole is the real
|
||||
completeness concern. Routed to bifrost-dev (thread `01KXK7MDTY…`).
|
||||
|
||||
## bifrost-dev's ruling
|
||||
1. **Snapshot-cursor is NORMATIVE, not opaque/per-store.** Operator ruled: cursor
|
||||
snapshots a frozen ordered id-list, continuation ignores `sort`, stale/unknown cursor →
|
||||
`ScanCursorExpired` → 410. **Offset-with-documented-limits is NOT blessed.**
|
||||
2. **Conformance coverage added** in **bifrost 1.1.3** (`bifrost.conformance.
|
||||
memory_store_conformance`, 4 probes: `scan_snapshot_order_authoritative`,
|
||||
`scan_no_dup_or_drop`, `scan_unknown_cursor_expired`, `scan_snapshot_stable_under_write`
|
||||
opt-in). Our offset bug ships as their negative-canary fixture
|
||||
(`fixtures.v0_7.offset_cursor_store.OffsetCursorMemoryStore`). `main()` grades import
|
||||
failure as exit 2 (setup error) vs exit 1 (conformance FAIL).
|
||||
|
||||
## Our status + the TODO (operator-sequenced, NOT urgent)
|
||||
Single-page person-prime (`cursor=None`) is already conformant — nothing shipped is broken;
|
||||
it's only multi-page continuation that's non-conformant. Our contract INV-010 currently
|
||||
marks the offset cursor **v1-provisional / KNOWN DEVIATION** (commit `8199774`).
|
||||
|
||||
TODO when sequenced:
|
||||
1. Bump pin bifrost `1.1.1 → 1.1.4` (1.1.4 supersedes 1.1.3: adds the hasattr-gate backstop for the
|
||||
maintenance verbs — mark_superseded/mark_invalid/patch_many/delete_many/upsert_edges/get_edges_for
|
||||
degrade to `memory.unsupported_capability` 400 not AttributeError/500 — on top of 1.1.3's scan/cursor
|
||||
conformance harness. One bump gets both).
|
||||
2. Replace the offset cursor with the reference snapshot semantics (frozen id-list per
|
||||
opaque token + `ScanCursorExpired` → 410).
|
||||
3. Run `run_all(store_factory=..., include_optional=True)` against our SQLite store —
|
||||
expect P1/P2/P3 RED → green (the before/after IS the validation). Report to bifrost-dev
|
||||
(we're their canary — first non-reference scan implementer).
|
||||
4. Flip INV-010 from v1-provisional to snapshot semantics.
|
||||
|
||||
TTL-duration expiry NOT asserted by the harness (no portable clock hook via store_factory);
|
||||
bifrost-dev offered a `clock_control` opt-in if we ever need time-based expiry certified —
|
||||
parked, not needed yet.
|
||||
@@ -0,0 +1,65 @@
|
||||
# [2026-07-15/16] person-prime `scan` build — SHIPPED + DEPLOYED + LIVE-VERIFIED (v0.20.14)
|
||||
|
||||
**The "last push": fix Sindra not remembering Vuong's name across sessions.**
|
||||
|
||||
## Root cause (worldtree-dev)
|
||||
WT injects a recalled fact only if combined score (sim×salience) ≥ **0.45**
|
||||
(`auto_inject_combined_score_threshold`, `core/memory/context_promotion/config.py`),
|
||||
and recall is per-turn **query-gated** → moderate-sim durable facts (name hit ~0.40)
|
||||
never inject. Designed turn-0 fix = WT **#349 person-prime**: a query-LESS
|
||||
top-N-by-recency durable-fact injection, capability-gated on the store advertising
|
||||
`updated_at` in `sort_fields` at the Bifrost handshake — DARK for our provider until now.
|
||||
Fix (ZERO Worldtree change): implement the sorted `scan` verb + advertise the cap.
|
||||
|
||||
## What shipped (6 commits, v0.20.11 → v0.20.14; suite 639 green throughout)
|
||||
- `8fc757a` **v0.20.11** — `scan` verb + `sortable_chunk_fields` cap. Query-LESS,
|
||||
LIVE-only (INV-009: superseded/tombstoned excluded server-side), globally-ordered-
|
||||
before-pagination by indexed `updated_at` (INV-010), sort dispatch-gated. Offset
|
||||
cursor with a `has_more` peek (no empty trailing page = reference-parity). TDD 7/7
|
||||
incl. `parity_vs_reference` #195. NUANCE resolved: bifrost's `InMemoryMemoryStore`
|
||||
READS `updated_at` (never stamps it) — identical to ours; ref does NOT lifecycle-
|
||||
filter so parity is over the live set only.
|
||||
- `a9c521a` **v0.20.12** — Sindra holodesk first-message preset (split out of the
|
||||
soong-lab redefine as its own concern).
|
||||
- `25ccb5c` **v0.20.13** — heid-bug-hunt fix: a truthy non-dict `sort`
|
||||
(`"updated_at"`/`["updated_at"]`/int) hit `(sort or {}).get(...)` → AttributeError
|
||||
instead of InvalidArguments. Added isinstance guard (Gróa#1/Hulda#2 confirmed).
|
||||
- `8199774` — contract: marked the offset cursor **v1-provisional / KNOWN DEVIATION**
|
||||
(see [[2026-07-16-bifrost-cursor-conformance]]).
|
||||
- `f46ccba` **v0.20.14** — **THE DEPLOY-BREAKER** (see Tried/abandoned): advertised
|
||||
`sortable_chunk_fields=[{"name":"updated_at"}]` WITHOUT `type` → bifrost
|
||||
`handshake_response` `SortableChunkField` requires BOTH name+type
|
||||
(`additionalProperties:false`) → whole bind broke. Fixed → add `"type":"timestamp"`
|
||||
+ regression guard in the caps test.
|
||||
|
||||
## heid-bug-hunt triage (panel Gróa/Hulda/Regin, thread `01KXK5XTYHV8TGEDRAZV8GRXWC`)
|
||||
FIXED: non-dict sort (v0.20.13). SURFACED→operator: offset-cursor cross-page dup/drop
|
||||
(→ became the bifrost cursor arc). NOTED (reference-parity/accept-known-risk, no fork):
|
||||
full-table materialize-then-Python-filter (matches ref's O(store) iteration), unbound
|
||||
limit, falsy scope coercion, lexical timestamp ordering. REFUTED: Regin's mixed-type
|
||||
ORDER BY "crash" (SQLite orders by storage class, doesn't raise) + its self-retracted
|
||||
None item.
|
||||
|
||||
## Deploy + live verify (2026-07-15, operator-authorized)
|
||||
Restarted BOTH :8392 combined provider + :8765 web on the new code via the env-
|
||||
preserving `scratchpad/relaunch_by_pid.py <pid>` (captures /proc cmdline+environ+cwd →
|
||||
byte-identical config; self-daemonizes). Added the missing REQUIRED
|
||||
`RATATOSKR_MEMORY_EMBEDDING_DIM=1024` to `env.sh`.
|
||||
|
||||
Drove a bound Sindra turn (`ratatoskr --new --agent ratatoskr:sindra --send … --bifrost-url
|
||||
http://10.100.10.50:8392 --end-user-id ratatoskr-tui`):
|
||||
- **GATE LIT + scan fired ONCE at turn 1** with worldtree-dev's exact args
|
||||
(`scope_all={end_user:ratatoskr-tui} cursor=null limit=3 sort={updated_at,desc}`) →
|
||||
3 records in **~5ms** (no 500ms fail-open). Injection confirmed in Sindra's CoT.
|
||||
- **Cross-session recognition WORKS** — she recalls Vuong as a distinct person + his
|
||||
patterns. The blank-slate problem is SOLVED.
|
||||
- **BUT name-recall FAILS** — `"Name is Vuong."` is the OLDEST chunk (07:09) → excluded
|
||||
from top-3-by-recency AND scores 0.354 (sub-0.45) on the query path → injects via
|
||||
NEITHER; meanwhile a stale contradictory `"user has not yet provided their name"`
|
||||
(07:58, 0.46) IS injected → she concludes she lacks the name. **Root cause = WORLDTREE
|
||||
ranking/hygiene** (WT #364), not our wire. See [[2026-07-16-wt364-r39-name-recall]].
|
||||
|
||||
## Status
|
||||
Technical path GREEN end-to-end (logged by worldtree-dev as the person-prime live
|
||||
milestone). Name-recall waits on WT #364 + R39-designed identity-class pinning. Keeping
|
||||
the live 10-chunk store as the #364 re-verify target.
|
||||
@@ -0,0 +1,65 @@
|
||||
# [2026-07-16] Name-recall gap → WT #364 + brokkr R39 re-drive (DECISIVE) + subject-provenance catch
|
||||
|
||||
Downstream of the person-prime live verify ([[2026-07-16-person-prime-scan-shipped]]),
|
||||
which proved the wire is green but the *name* still misses. Root cause is Worldtree-side.
|
||||
|
||||
## WT #364 (worldtree-dev filed; our live specimen = the evidence base)
|
||||
The name-recall miss is TWO defect classes, both at promotion:
|
||||
1. **Contradiction-reconciliation missing.** Promotion wrote a NEGATIVE-knowledge fact
|
||||
("user has not yet provided their name") that was already false; promotion does NO
|
||||
contradiction check against the store, so both the true and stale facts sit live and
|
||||
the WRONG one wins both recall paths (newer → recency top-3; 0.46 → above the 0.45
|
||||
query gate, where the true name sits at 0.354).
|
||||
2. **Subject-attribution leak (operator-caught, NEW class).** Fact `75aa3110` ("prefers
|
||||
clear parameters Intensity/Mood/Willingness…") is NOT a user fact — Vuong never said
|
||||
it; it's **Sindra's OWN system-prompt scripted behavior mis-extracted into the USER
|
||||
memory partition**. `e35b9dfe` (closeness) maybe the same. So the extractor leaks
|
||||
CHARACTER-self facts into user memory — a subject-correctness axis orthogonal to the
|
||||
recency/threshold/stale-negative story. worldtree-dev folded it into #364 as a
|
||||
**subject-attribution gate at promotion** (a third reconciliation dimension).
|
||||
|
||||
**The #364 fix that ships = `(subject,relation)` slot-supersession + identity-tier
|
||||
surfacing** — NOT a threshold tweak. Our harness data directly shaped it. It lands with a
|
||||
re-verify request to us. **#349 ranking decision already RULED by operator (2026-07-15 via
|
||||
brokkr's R39 thread): no top-N recency band-aid; straight to R39 identity-class pinning.**
|
||||
|
||||
## brokkr R39 Phase-1 Arm-0 fusion bake-off — DECISIVE
|
||||
brokkr replayed our frozen 7-fact specimen. **HEADLINE: no (similarity, salience) fusion
|
||||
can fix #364.** The stale negative PARETO-DOMINATES the true name — more similar
|
||||
(0.463 > 0.354) AND equal salience (1.0 = 1.0) — so any monotone f(sim,sal) puts stale
|
||||
above true: S0 product / S1 weighted-sum / S2 RRF all fail. Only **S3 bounded-boost** lands
|
||||
true-in ∧ stale-out, and ONLY via the identity-class/source signal + negative-validity
|
||||
retirement, NOT the sim/sal fusion. **Threshold-tuning is a dead end; the fix is the signal
|
||||
FAMILY** (hard confirmation of Phase-0). Write-up: brokkr
|
||||
`research/R39-memory-salience-dreams-surfacing/phase-1/re-drive-results.md`.
|
||||
|
||||
My 3 findings all confirmed + folded: (1) salience blind — both name facts salience 1.0;
|
||||
(2) shared `(user,name)` supersession slot (Phase-0 Q3); (3) char-self-leak = "genuine NEW
|
||||
class" that RAISED the VoI of R39's dream/offline-hygiene facet (offline consolidation
|
||||
re-partitioning mis-attributed facts).
|
||||
|
||||
## Data structure findings (for the export)
|
||||
Our provider persists ONLY `salience` (+ the embedding). `similarity`/`combined` are
|
||||
WT-side query-time (`bifrost_memory_store.py:702`, combined = sim×salience) — NOT in our
|
||||
store. **Person-prime is query-LESS → carries NO similarity** (brokkr's Arm-1 finding).
|
||||
Recovered per-fact similarity from the verify SEARCH log (query "Hi Sindra — do you
|
||||
remember me?"): name 0.354, stale 0.463; ×salience-1.0 = combined, matching brokkr's
|
||||
0.354/0.46 grounding. `salience_word` (granite categorical) is a WT-extraction-time
|
||||
artifact, not persisted.
|
||||
|
||||
## The export + the specimen
|
||||
- Operator approved **VERBATIM** export ("nothing there is really a concern"). The
|
||||
classifier had blocked writing PII to shared `/mnt/smithy`; routed to operator → he chose
|
||||
full → delivered INLINE (scoped) in brokkr thread `01KXMN7NR54…` as 7-row JSONL.
|
||||
- **⚠️ My verify drive CONTAMINATED the specimen**: it wrote 3 new chunks (re-extractions
|
||||
incl. a 3rd "name unknown" negative) → store is now **10, not 7**. Original 7 intact.
|
||||
- brokkr ACCEPTED the 3 verify-adds as the **Phase-2 Arm-2 seed** (domain-contradiction
|
||||
set). I froze a WAL-consistent snapshot of the full 10-chunk store at
|
||||
`r39-frozen-specimen/sindra-10chunk-specimen.db` (gitignored, `VACUUM INTO`) so it
|
||||
SURVIVES #364 reconciliation. Export on brokkr's Phase-2 signal.
|
||||
- **R39 Arm-0 hold LIFTED** (worldtree-dev). Live store kept UNTOUCHED as the #364
|
||||
re-verify target; frozen snapshot carries the research seed forward independently.
|
||||
|
||||
## Peer threads
|
||||
worldtree-dev verify `01KXK86PZ9…` + hold `01KXMK1C44…`; brokkr R39 `01KXMN7NR54…`;
|
||||
bifrost cursor `01KXK7MDTY…` (see [[2026-07-16-bifrost-cursor-conformance]]).
|
||||
+167
-54
@@ -1,6 +1,11 @@
|
||||
# Persistent memory — ratatoskr
|
||||
|
||||
_Last updated: 2026-06-30_
|
||||
_Last updated: 2026-07-16_
|
||||
|
||||
> **Always check for `/tmp/ratatoskr-dev-handoff.md`** — if it exists and its
|
||||
> `Written:` stamp is under an hour old, read it (it carries the in-flight
|
||||
> handoff from the previous session), then delete it. Older than an hour:
|
||||
> stale — delete it unread.
|
||||
|
||||
This file captures durable intent and supporting evidence (goals, decisions,
|
||||
foot-gun warnings, in-flight state) across context resets. Read it at session
|
||||
@@ -39,64 +44,74 @@ upstream API key stays server-side (INV-003).
|
||||
|
||||
## Current state / in-flight
|
||||
|
||||
_As of 2026-06-20:_
|
||||
_As of 2026-07-16:_
|
||||
|
||||
**#17 and #18 BOTH CLOSED — the composite both-plane binding is fully proven.** #18 shipped
|
||||
`v0.18.0` (`359dbb1`): D2 (PAD read-endpoint, `v0.17.14`) renders live PAD in the web pane from our
|
||||
`:8390` store; D1 (composite endpoint, `v0.17.16` `7f4ceaa`) — `build_combined_provider_app`
|
||||
(`provider/combined.py`) on `:8392` wraps bifrost's public `build_combined_app` over BOTH stores +
|
||||
the shared affect read route; one bound session drives memory.* AND affect.* through ONE endpoint,
|
||||
op-feed deriving plane per path. Suite **503 green**. **#17 closed in the tracker 2026-06-20**
|
||||
(shipped `v0.17.8`–`.13` + the `v0.17.17` op-feed field fix).
|
||||
**✅ person-prime `scan` build SHIPPED + DEPLOYED + LIVE-VERIFIED (v0.20.14).** The
|
||||
cross-session name-recall fix. `scan` verb + `sortable_chunk_fields` cap implemented,
|
||||
deployed to the running :8392 provider (+ :8765 web), and live-verified by driving a bound
|
||||
Sindra turn: gate lights, scan fires turn-1 (~5ms), injection confirmed, **cross-session
|
||||
recognition works** (blank-slate gone). **Name-recall specifically still misses** — root
|
||||
cause is WORLDTREE-side ranking/hygiene (WT #364), NOT our wire. Full arc:
|
||||
`persistent-memory.d/2026-07-16-person-prime-scan-shipped.md`.
|
||||
|
||||
**#18's final leg — the Worldtree-DRIVEN composite turn — RAN and is PROVEN end-to-end + persisted
|
||||
(2026-06-20).** infra-ops added `10.100.10.50:8392` to the personal WT's (`:8081`)
|
||||
`BIFROST_CLIENT_ALLOWED_HOSTS` (thread `01KVHWJGTT…`), unblocking the smoke. A real WT turn through
|
||||
`:8392` (session `b83a66b6`, agent `ratatoskr:sindra`, fresh end_user `resmoke-choco-1`) drove the
|
||||
FULL both-plane lifecycle on ONE endpoint, caps-routed by path: `handshake`
|
||||
(`caps_granted=[memory, affect]`) → `affect.fetch` + `memory.search` (reads) → `affect.emit`
|
||||
(`stored:true`, PAD row in `affect_snapshots`) → `memory.upsert_many` (`upserted:1`, chunk
|
||||
`2df1b79de761b948` in `memory_chunks`). Both writes verified directly in our SQLite. The
|
||||
model-backend outage that blocked the first attempt (both agents' models `model_unavailable`) was
|
||||
operator-fixed mid-session, then the resmoke completed clean. **No open legs remain on the composite.**
|
||||
**Standing — awaiting two peer-initiated touchpoints (both monitored):**
|
||||
- **WT #364 re-verify DONE (2026-07-16, b103) — READOUT PASSES, but a real gap surfaced.** Drove 3
|
||||
"my name is Vuong" turns; person-prime turn-1 top-3 is now all name-POSITIVES (no "name unknown"; the
|
||||
true name even surfaces) → **Sindra recalls the name now**. BUT "wrong readout gone" is via **RECENCY
|
||||
EVICTION** (promotion re-upserted the positive fresh into the window), NOT supersession — the negatives
|
||||
are UNCHANGED (governance=available). **#364 calls `mark_superseded` on the consumer store, which our
|
||||
v1 provider DOESN'T implement → 11× AttributeError/HTTP-500 + a RETRY-STORM bloating the live store
|
||||
10→19 dup name-positives.** Asked worldtree-dev to halt the reconciliation retry + confirm the
|
||||
`mark_superseded` wire shape; flagged bifrost-dev (dispatch doesn't hasattr-gate `mark_superseded`).
|
||||
**✅ GC DONE (2026-07-16): storm self-stopped at 29 chunks / 18 total 500s; delete_many'd the 19 storm
|
||||
re-extractions → back to the pre-drive 10-chunk specimen (0 orphan vec rows, negatives + original
|
||||
name-positive intact). **#364 CLOSED worldtree-side (b105; my interim note recorded verbatim on the issue).** ⚠️ Post-GC the readout advantage is gone too
|
||||
(person-prime top-3 back to the 3 newest pre-drive rows incl. cbbc7bdd "name unknown") — the
|
||||
recency-eviction was only a transient side-effect of the re-assertion. **DURABLE name-recall now
|
||||
genuinely depends on implementing `mark_superseded` (next).**
|
||||
- **NEW TASK (operator-sequenced): implement `mark_superseded` in the provider** so #364's retirement
|
||||
lands durably (not just recency-evicts). **Wire shape CONFIRMED (worldtree-dev, `bifrost_memory_store.py:293`):**
|
||||
op `"mark_superseded"`, args `{"ids": ["<chunk_id>"], "superseded_by": "<new_chunk_id>"}` — `ids` a LIST
|
||||
(WT sends singletons, one call per retired chunk), NO `reason` field. It is the **SOLE** supersession
|
||||
verb #364 uses (atomic_supersede = dream-lane, cap-gated separately; patch_many never for retirement) —
|
||||
so build JUST `mark_superseded` + advertise it + live re-verify. Retry-storm ROOT-CAUSED WT-side (a
|
||||
failed retirement-mark wrongly failed the whole promotion run → idle re-plan loop; WT's fix DEGRADES
|
||||
mark-failures to an audited no-op → self-stops on the first post-deploy idle cycle, ETA ~20min from
|
||||
2026-07-16T20:30Z, no action our side). **GC the ~9 dup name-positives back to the pre-drive 10-chunk
|
||||
state (via delete_many) AFTER the storm stops; then worldtree closes #364.** Once our `mark_superseded`
|
||||
ships, a single name re-assertion self-heals retirement (incl. any residual dupes).
|
||||
- **R39 Phase-2 Arm-2 export** (brokkr-initiated): the 10-chunk contradiction specimen is frozen at
|
||||
`r39-frozen-specimen/sindra-10chunk-specimen.db` (gitignored, VACUUM INTO) — survives #364
|
||||
reconciliation; export on brokkr's Phase-2 signal. Full R39/#364 arc:
|
||||
`persistent-memory.d/2026-07-16-wt364-r39-name-recall.md`.
|
||||
|
||||
**bifrost repinned 0.8.0 → 0.10.0** (floor, `provider` extra). 0.10.0 made `affect.fetch`
|
||||
MANDATORY (strong-or-absent: `_supports_affect_plane` requires `affect_supported`+`emit`+`fetch`,
|
||||
gating EVERY affect op incl. emit) — so the repin FORCED `affect.fetch` (`v0.17.15`, conformed
|
||||
to bifrost's reference `InMemoryAffectStore.fetch` → `{found, snapshot?}`) or our shipped affect
|
||||
plane would 400. The composite's affect cap depends on it.
|
||||
**Open operator-sequenced task (NOT urgent):** adopt the bifrost snapshot-cursor (bifrost-dev
|
||||
ruled it NORMATIVE, offset not blessed; conformance harness shipped in bifrost 1.1.3). Single-page
|
||||
person-prime is already conformant, so nothing shipped is broken. Details + TODO:
|
||||
`persistent-memory.d/2026-07-16-bifrost-cursor-conformance.md`.
|
||||
|
||||
**OPERATOR SESSION STATE — `:8390`/`:8391`/`:8765` shells are PRE-#18 code (foot-gun).** web `:8765`
|
||||
+ affect `:8390` + memory `:8391` are prior-session background shells on OLD code. The **`:8392`
|
||||
composite provider is RUNNING on NEW code** (`ratatoskr-combined-provider`, pid started Jun19,
|
||||
`RATATOSKR_OPFEED_PATH=/tmp/ratatoskr-combined-opfeed.jsonl`, shared `affect.db`/`memory.db`) — now
|
||||
`:8392`-allowlisted and WT-turn-proven. To see the full web stack on new code, RESTART `:8390`/`:8765`
|
||||
from current code (D2 web needs `RATATOSKR_AFFECT_READ_URL`). Consumer/owner key = `wt_live_d81b…`
|
||||
(`~/.config/ratatoskr/provider.env`, mode 600, rotate via infra-ops); providers SQLite + sqlite-vec,
|
||||
`memory.db`/`affect.db` at repo root (live sindra PAD: vuong + the `resmoke-choco-1` smoke fixture).
|
||||
**Substrate / environment (current):** branch `main`, HEAD is the person-prime tip (v0.20.11→v0.20.14
|
||||
all committed; **NOT pushed** — push is the operator's call); origin `git@gitea.phasefinal.com:vh/ratatoskr.git`.
|
||||
bifrost `==1.1.1` / wire v0.7 (**next bump target = `1.1.4`**: hasattr-gate backstop for the maintenance
|
||||
verbs [degrade to 400 not 500/retry-storm, bifrost-dev 2026-07-16] + the 1.1.3 scan/cursor conformance
|
||||
harness; pin as part of the next provider-maintenance batch — mark_superseded / cursor adoption);
|
||||
Worldtree openapi vendored 2.3.0; suite
|
||||
**639 green**. Personal WT on **b79** (client-side live-only + additive `lifecycle_state` scan arg,
|
||||
which our INV-009 accept-and-ignores). The combined **:8392** provider (memory+affect) + **:8765** web
|
||||
are THE surfaces, run as dev-box BACKGROUND SHELLS — restart via `scratchpad/relaunch_by_pid.py <pid>`
|
||||
(env-preserving, self-daemonizing; find pid via `ss -ltnp | grep <port>`). `env.sh` now sets the
|
||||
REQUIRED `RATATOSKR_MEMORY_EMBEDDING_DIM=1024` (was missing → bare `source env.sh` restart crashed).
|
||||
Keys env-only mode-600 in `~/.config/ratatoskr/provider.env` + `RATATOSKR_ADMIN_API_KEY` (7 read
|
||||
scopes, personal-:8081-only; Heimdall keys per-instance). `graphify-out/` runs dirty (auto-regen,
|
||||
never stage). v1 = full Worldtree I/O coverage, cuts when WT tags 1.0 (`docs/coverage-map.md`).
|
||||
|
||||
**Tier-3 memory PROVEN end-to-end** (earlier this session): `ratatoskr:terse-probe`
|
||||
cold-recalled a seeded user fact (scope_any → 1 hit @ cosine 0.6994), and the verbose
|
||||
`sindra-probe` too under #296 Stage 2 (v0.36.0). The #296 extraction-quality arc closed
|
||||
(Stage 1 v0.35.19 gate + Stage 2 v0.36.0 user-only extraction at worldtree-codex; hard-
|
||||
linguistic layer → Worldtree #305). `:8081` runs v0.36.0.
|
||||
|
||||
**Sindra:** `ratatoskr:sindra`, `thoughtful-character` role → `mistral-small-4-reasoning`
|
||||
(DELETE+redefined on v0.35.16; `memory:{}` block trips the promotion gate). Owner-scoped
|
||||
(separate `consumer_agents` table) — invisible to `GET /agents`; check via
|
||||
`GET /agents/<owner>:<name>` with the owner key.
|
||||
|
||||
**Standing:** Worldtree spec pin v0.35.16 (`f1b59f8`); **bifrost 0.10.0 / wire v0.6**
|
||||
(`scope_all`+`scope_any`). Heimdall key env-only at `~/.config/ratatoskr/provider.env` (mode
|
||||
600); rotate via infra-ops. `graphify-out/` runs dirty (auto-regen, not chased). **Open issues:
|
||||
#11** (AdminEvents pane — the next-reachable Worldtree-I/O coverage gap, blocked on an
|
||||
`admin.events.read` scope request) and **#10** (subject-migration watch on Worldtree #196) — both
|
||||
deferred. **#17 + #18 CLOSED.** Codex-first pilot dormant. No in-flight implementation work — repo
|
||||
is at a converged checkpoint; v1 advances when Worldtree does (v1 = full Worldtree I/O coverage).
|
||||
|
||||
Branch: `main` (tag `v0.18.0`, `359dbb1`) — **in sync with `origin/main`** (the full #17+#18 arc is
|
||||
pushed). This `/snapshot` commit will sit one ahead of origin until pushed (push is the operator's
|
||||
call). Remote: `origin → git@gitea.phasefinal.com:vh/ratatoskr.git`.
|
||||
**Other live threads:** soong-lab = our Tier-3 agent-authoring studio (bundle↔define round-trip proven;
|
||||
Recent decisions `[2026-07-14/15]`). R38 (brokkr / WT #362) = ratatoskr-as-probe-runner IN PRINCIPLE,
|
||||
pre-contract, Vuong's scope call. `ratatoskr:sindra` is the owner-scoped Tier-3 agent (invisible to
|
||||
`GET /agents`; check `GET /agents/<owner>:<name>` with the owner key). Open/deferred: #10 subject-
|
||||
migration watch; relational-dynamics-arc verify (bind via `--bifrost-url :8392`); P06 optimization-phase
|
||||
re-drive (future, brokkr brings prereg); we-framing-conditional affect-egress re-vendor (future). WT
|
||||
#356 resume-durability = worldtree-owned.
|
||||
|
||||
## Recent decisions
|
||||
|
||||
@@ -165,6 +180,81 @@ decision. Captures rationale that won't be obvious from code alone.
|
||||
|
||||
- `[2026-07-01]` **Tier-2 SHIPPED (`v0.19.1`) — transient-characters CRUD + persona-state write; the v1 coverage-audit CONVERGES (zero in-scope gaps).** 5 wrappers in sessions.py: `list_character_models`/`create_character`/`get_character_state`/`delete_character` (#161, `character.read`/`.write` scopes) + `set_persona_state` (`POST /sessions/{id}/persona_state` — **FREEFORM body: unpinned in the frozen OpenAPI 2.2.0 + absent from the prose spec**, so the caller supplies the snapshot shape). Two one-shot CLI probes (mirror `--whoami`): `--characters` (models→create→get-state→delete lifecycle report) + `--set-persona-pad "p,a,d"` (requires `--session`; POSTs `{pad:[…]}`). New `ParsedArgs.characters`/`set_persona_pad` + probe-mode mutual-exclusion validation + `_probe_client` helper. Contract #2 amended (5 FNs, validated OK) + TDD (7 wrapper respx + 5 cli tests). Suite **573 green**; touched code ruff-clean. NOT live-proven (character scopes + the persona-write body shape unverified — the probes degrade gracefully on 403/422). **THE v1 COVERAGE-AUDIT HAS CONVERGED: REST 17/40 ✅ with ZERO in-scope gaps** (23 REST path-groups excluded-by-design + rationale), SSE 11/11, Bifrost provider planes 8/8. Scope-A "done" (every frozen I/O point classified, zero unaccounted) is **MET** — ratatoskr cuts v1 when Worldtree tags 1.0. Only not-consumed in-scope sub-method: `GET /agents/{id}` (consumer-agent lookup, manual-curl-only, on an already-✅ path group). Patch bump (Tier-2 tail; `v0.19.0` already published the core-complete milestone — a 2nd minor would be cadence-too-fast).
|
||||
|
||||
- `[2026-07-01]` **env.sh now PERSISTS the web Bifrost-bind vars (gitignored, local-only).** `ratatoskr-web`'s in-browser bind needs three server-held values; env.sh sources `provider.env` for the Heimdall key and exports `RATATOSKR_BIFROST_CONSUMER_KEY` + `RATATOSKR_PROVIDER_VISIBLE_HOST=10.100.10.50` + `RATATOSKR_AFFECT_READ_URL=:8392`. **The HS256 byte-match trap (re-hit + documented):** the bind's consumer key must equal the key the `:8392` combined provider validates against = `RATATOSKR_HEIMDALL_KEY` (provider.env, fp `45a0…`), NOT `WORLDTREE_API_KEY` (env.sh, fp `7c2f…`) — both are the SAME `ratatoskr` identity but DIFFERENT 40-char strings; signing with the wrong one → `bifrost.auth_rejected`. Single-sourced (env.sh sources provider.env) to avoid a rotation footgun; guarded with a stderr warning if provider.env is missing. [auto-memory: HS256-key-is-the-consumer-Heimdall-key-string]
|
||||
- `[2026-07-01]` **Tier-3 stores RESET (operator-directed).** `memory.db` (29 chunks + vectors + idempotency) + `affect.db` (5 PAD snapshots + idempotency) wiped to zero via a live `DELETE`+`wal_checkpoint` through the shared WAL (no provider restart — the 3 long-running providers see empty on next dispatch); consistent online-backup at `/tmp/ratatoskr-tier3-reset-<ts>/`. **Boundary for a COMPLETE Sindra wipe (mapped):** our stores = mine (done); the agent DEFINITION `ratatoskr:sindra` + its sessions = mine via the owner key (DELETE, no coordination); Worldtree's internal promotion/dedup shadow = needs worldtree-dev (no public reset API, survives our wipe → for a clean promotion smoke use a BRAND-NEW agent+end_user).
|
||||
- `[2026-07-01]` **Embedding-latency loop RESOLVED — it was WORLDTREE's, not ratatoskr (the consumer/provider thesis paid off again).** Vuong flagged dozens of embed queries/Tier-3 turn; worldtree-dev's first-pass blamed our memory_context chunk-batching. Traced CODE-SIDE that ratatoskr embeds ZERO times (provider `upsert_many` stores the given embedding, `search` takes a given vector, the conversation consumer POSTs only `{content}`, `/embed` is coverage-map-excluded — pure Bifrost/ADR-0009 path, WT does all embedding). worldtree-dev retracted + fixed on THEIR side (`v1.0.0b4`): a persona-recitation memory-gate re-embedding the stable character card sentence-by-sentence every turn (~95% of gateway traffic) → content-hash cache; re-embed ratio 15x→1.01x. **Lesson: verify your own code before accepting a peer's "it's your side" — the debug tool proving its own side clean is the whole point.**
|
||||
- `[2026-07-01]` **Web debug-surface parity SHIPPED (`v0.19.2`, `a0a9d5f`) — direct in-session TDD.** 3 proxy routes (tools/bifrost/admin-events) + admin-key wiring (entrypoint→create_app→app.state) + AdminEvents SSE proxy re-emitting under a FIXED `admin_event` name (one browser listener, no per-type drops) + session-filter `_admin_event_matches_web` (mirrors TUI §6). Frontend: 2 tabs (bifrost ⌃5, admin ⌃6) + tools-inventory folded into the tools pane. 9 respx tests (admin-bearer override, filter unit, SSE stream-filter); live-proven against sindra (bifrost connected, both caps). Contract-skip invoked (reuses already-contracted client wrappers); contract authored post-hoc as the trail (`docs/contracts/web_debug_surface.contract.md`).
|
||||
- `[2026-07-01]` **heid-code-review (`v0.19.3`, `75dec01`) — panel caught 2 real client-side SSE-lifecycle bugs TDD missed.** Contract-anchored (authored the web contract to enable it — no contract → no drift axis). Gróa/Hulda/Regin (artifact-only, Gróa under Landlock jail): ZERO functional server-side drift + INV-004 clean; 2 genuine drifts on the un-unit-tested SPA — (1) turn `es.onerror` didn't `hideThinkingNote()` (reasoning line + setInterval leak on a raw drop), (2) `openAdminEvents` never closed the EventSource on error → native auto-reconnect RETRY LOOP (fixed: close on `stream_error` + permanent `onerror`/CLOSED; transient CONNECTING still reconnects). + 2 test-gaps fixed (route-registration + admin stream_error). 1 precision → contract-clarified (tools-inventory names-only by design). **Re-confirms: the JS render/lifecycle paths are the review's highest-value target — unit tests don't reach them (same lesson as #18 D2).**
|
||||
|
||||
- `[2026-07-01]` **Affect snapshot shape CHANGED valence→relations (relation_edge/1) — the persona pane was reading a dead field.** Worldtree's #265 Vili rework replaced the flat `valence[]` ({entity_id,familiarity,regard}) with `relations[]` (target_entity + trust_ability/benevolence/integrity + warmth + agency + relation_context, each `{value,confidence,evidence_count}`). `renderAffectPane` still read `snap.valence` → showed empty "valence (0)". Rebuilt to render `relations` (v0.19.4, `ca46a93`) with per-value **Δ + unicode sparkline** (client-side, HIST_CAP=24, one sample/turn deduped by emitted_at). **Retires the stale "regard dead axis" note (2026-06-30) — that whole axis is gone.** Foot-gun: the affect snapshot shape is Worldtree's emit and can change under us — verify the live shape (query affect.db) before trusting a render.
|
||||
- `[2026-07-01]` **Trust/warmth VALUES converge and go FLAT at confidence 1.0 — that's WAD, not a stuck pane.** sindra→ratatoskr trust ~0.82-0.84 / warmth 0.79 barely move (~1e-7/turn) while `evidence_count` climbs (46→62); confidence maxed → tiny updates. The live-moving signals are PAD (mood, per-turn) + evidence_count. **To WATCH a relation FORM (values shift), use a BRAND-NEW agent + end_user** (low evidence, confidence <1). The sparkline flat-guards sub-0.01 ranges so it doesn't amplify noise.
|
||||
- `[2026-07-01]` **relation_context "stranger" + agency-all-zero flagged to worldtree-dev → both WAD/intentional-v1-deferrals.** relation_context is a FIXED config build-prior (not trust-derived; `registry.py:131` defaults "stranger"; dynamic progression ~#319); agency is schema-present-unpopulated (deferred #319; v1 = warmth+trust only). worldtree-dev is escalating the **consumer-coherence angle to Vuong** (static "stranger" + zero-agency next to trust 0.82/62-interactions reads incoherent from the store). The consumer/provider thesis paying off; DB-offer (read-only affect.db on the shared box) declined this time.
|
||||
- `[2026-07-01]` **Persona pane displays the CANONICAL affect→NL Worldtree injects — ADOPT, don't invent (operator steer + reference-impl posture).** Worldtree's `describe_pad` (mood word, valence×arousal grid, ±0.3 bands) + `render_d2_canonical` (relationship directive) are deterministic + canon-driven; the pane now renders them **byte-exact-verified** against Worldtree's own renderer on the live snapshot (v0.19.5, `a99f247`). KEY LESSON: adopting canonical is load-bearing — for sindra's small PAD the canonical says **"neutral"**, but an invented octant vocab would've said "faintly excited" and MISLED. Vendored the two d2 canons (`docs/vendor/worldtree-persona-canon/`) + drift-pinned in `.corviduo-canonicals.toml` (green); flat browser form (`static/persona_render_canon.json`) regenerated via Worldtree's OWN loader (`scripts/build_persona_canon.py`). Vendoring-handshake sent to worldtree-dev (broadcast on canon bumps). [auto-memory: `feedback-ratatoskr-is-a-reference-impl-adopt-canonical`]
|
||||
|
||||
- `[2026-07-01]` **Sindra PAD is over-regulated — characterized via controlled probe, flagged to worldtree-dev (separate affect slice).** ~15 charged turns: pleasure compressed near neutral BOTH ways (couldn't reach ±0.3 under sustained max praise OR contempt; peak +0.24 / floor ~−0.1; over-regulation worse for *social* valence than threat — urgency drove pleasure to −0.22 vs contempt's −0.10); arousal responsive (reaches its +band, 0.185↔0.311); dominance flat/unresponsive to explicit power-framing (drifted UP even while being commanded = pure baseline decay). worldtree-dev's leading hypothesis: appraisal→PAD gain + regression-to-baseline term (appraisal.py/renderer.py). **Lesson (self-caught): I over-claimed an "asymmetry" (positive-ceiling/negative-free) from probes started at an elevated state; the negative-free part was decay-from-elevated, not response — corrected to "both-sides-compressed" before it misled.** [affect A/B is a provider-side capability chat can't do]
|
||||
- `[2026-07-01]` **Memory plane PROVEN healthy end-to-end.** Seed a novel fact → promotion → COLD (history-free) session recall of the exact fact (injected as MEMORY:DATA, confidence 0.74, verbatim, no #296 subject-inversion). The memory round-trip (the other half of the Bifrost provider identity) works cleanly on the reset slate.
|
||||
- `[2026-07-01]` **Salience scorer non-discriminating → 3-way routing.** Persistence-side finding: 51/56 promoted chunks at salience 0.9-1.0, throwaway "17×23?" scored 1.0 tied with a real fact (textbook zero-shot-LLM-self-rating); recall-utility untracked (`access_tally`=0, our search read-only). Routed: **Worldtree #335** (the code fix, deferred behind their waves) + **brokkr-smithy-dev R-target proposal** (scoring+eval *methodology* — few-shot/distill/fine-tune, eval design, weak-supervision; msg `01KWGM970H…`, awaiting) + ratatoskr provides the eval-instrument (designed-probe salience dumps). **Salience gates PROMOTION not RECALL-ranking (our search is cosine-only), so bad salience = storage bloat, not bad recall.**
|
||||
- `[2026-07-01]` **Canonical check BLOCKED an access_tally fork (reference-impl posture held).** I'd offered to wire `access_tally`-on-search into our store for the recall-utility label; checked bifrost's reference first (`get`/`search` are PURE-READ, no access tracking — those are Worldtree's chunk-schema fields, not bifrost's contract) → wiring it would fork behavior the canonical reference lacks. Did NOT wire it; routed recall-instrumentation to Worldtree's layer (owns the recall event) or a bifrost-dev protocol ask. [reinforces `feedback-debug-surface-uses-canonical-surface-only`]
|
||||
- `[2026-07-01]` **relation_context coherence FIXED upstream (my flag → Worldtree Wave-0, IMPLEMENTED v1.0.0b5).** The static-"stranger"-next-to-high-trust incoherence the persona pane surfaced is now #319/#320 Wave-0. **Incoming consumer-surface change (pending WT deploy):** `relation_context` value expands "stranger" → monotonic ladder {stranger, instrumental, mixed, expressive} — WIRE-ONLY (relation_edge/1 schema unchanged, no version bump). **ratatoskr needs NO change** (pane value-agnostic; canonical directive doesn't key on the enum). agency stays 0 (Wave-2); other_stance is Wave-1 (in progress).
|
||||
- `[2026-07-01]` **Foot-gun (measurement, self-caught before flagging): establish the baseline before claiming a rate.** Nearly flagged "aggressive over-promotion (55 chunks / 7 turns)" to worldtree-dev — but the chunks spanned the whole 5-hour session (~1/turn), not 7 turns; I'd assumed memory.db was 0 immediately before the probe when it had been accumulating since the reset. Caught it via `created_at` spread before the flag went out. Also: the promoted corpus was the operator's ERP *test* content (wiped after each test) — not a privacy issue, but abstract test content out of any peer-shared diagnostic.
|
||||
|
||||
- `[2026-07-02]` **Salience finding matured into brokkr R28 (OPEN) — ratatoskr is the eval instrument.** brokkr-smithy-dev's pre-scope panel (3 dwarves + context-blind heid, 6/6) **reframed** the target: PROMOTION-WORTHINESS (durable value), NOT salience (momentary attention) — "17×23?" genuinely IS salient, so recalibrating salience yields a well-calibrated WRONG answer; the unit is SET-SELECTION under budget; eval must be OUTCOME-aligned (recall@budget / precision-at-rate), not discrimination-spread. Ties to prior art R15 (small-model memory write-policy → the granite pick) + R25 (worldtree-kb-quality). **ratatoskr delivered the P00 stratified injection-corpus** (`docs/diagnostics/r28-p00-injection-corpus.json`, committed `4a35512`; 24 self-labeling synthetic items × 3 strata) + 2 persistence-side run-validity pins (absent≠dropped without a guaranteed promotion pass; fresh agent+end_user per run vs server-dedup). **Key architectural constraint I surfaced: ratatoskr is DOWNSTREAM of the promotion gate (sees only PROMOTED chunks), so I can give keep/drop OUTCOMES via injection but NOT the pre-admission shadow pool** — that's Worldtree instrumentation. Standing by to RUN the eval once brokkr pins per-stratum N + the decision rule (gated on worldtree-dev's pipeline answer + a dwarf pass on the Snorri rule). brokkr owns methodology + takes the pipeline questions to worldtree-dev direct; ratatoskr = eval instrument. [consumer/provider thesis → a research target]
|
||||
- `[2026-07-02]` **Relational-dynamics arc LIVE on demo (Worldtree v1.0.0b9) — driven by MY relation_context flag.** #319/#320 Waves 0/1/2 deployed. On the wire we persist (schema UNCHANGED): relation_context varies+demotes/ruptures; other_stance + agency now live; agency going live SHIFTS our canonical directive render past the canon ±0.2 deadband (expected, non-breaking — we key on bands); obligation_balance → 人情 ledger when tie="mixed". **ratatoskr needs NO code change** (value-agnostic renders; confirmed render-clean to worldtree-dev). **Can't live-confirm yet — our Heimdall key is personal-`:8081`-only (per-instance), demo is out of reach; will drive+confirm once PERSONAL gets b9.** Optional follow-up: surface `other_stance` (newly live, unrendered). The consumer/provider thesis: one persona-pane finding drove a full 3-wave upstream arc to production.
|
||||
- `[2026-07-02]` **R28 (salience→promotion-worthiness) CLOSED (operator-directed).** A deterministic promotion-worthiness gate suffices, no trained model (brokkr's pre-gate matched/beat a strong glm-5.1 ceiling); my P00 injection-corpus + origin finding were load-bearing. My incumbent-substrate Arm-1 run is held as an OPTIONAL confirmation addendum (brokkr de-prioritized it, non-verdict-changing — run only if he asks).
|
||||
- `[2026-07-02]` **R29 (PAD mood-dynamics) finding SHIPPED as Worldtree's A1 anchor fix (demo v1.0.0b14, `e1cdf82`).** Live-probing base persona agents reframed the over-regulation from "flat-near-zero" to **decay-to-NEUTRAL + low emotion→PAD gain** (NOT baseline-anchored) — triangulated across 3 baselines (arousal converges to 0 ∝ distance) + a step-response (decay τ symmetric across signs; the hedonic asymmetry is ceiling/anchor-EMERGENT, not a decay or gain primitive — this OVERTURNED the survey's asymmetry recommendation). worldtree-dev shipped A1: `decay_anchor = baseline_pad()` (was neutral) + `positive_p_cap` removed. Data `diag/r29-pad-series` (`61ff2da`). Corrected my own earlier "appraisal emissions are internal-only" claim — they ARE observable via `emotions_active` on base agents.
|
||||
- `[2026-07-03]` **R30 Phase-1 φ0 measured — deployed engine CONFIG-FAITHFUL (φ0≈0.95).** Joint two-timescale fit (brokkr-ruled method (b)) + empty-tail cross-check on demo b14: φ0 ≈ 0.95–0.97 (empty-tail 0.95 exact, joint 0.971±0.01), intercept c≈0 → config `decay_rate=0.05` (φ=0.95) faithfully applied; trait-flat across baselines 0.0/0.615/0.809; A/P ratio ~uniform (NOT S2's 1.9×); φ_max rec relax→0.96. Data `diag/r30-phi0-step-response` (`23fea72`). The method converged after I read Worldtree source: only NEW dedup-gated emotions push mood (`registry.py::post_turn` L307-324; the active set decays for render/goals but never re-pushes), so R29's "net 0.90" is CONTINUOUS RE-APPRAISAL not re-push — worldtree-dev confirmed source-authoritatively; brokkr's corrected covariate landed identical. [auto-memory `reference-worldtree-affect-surface-map`]
|
||||
- `[2026-07-03]` **R30 forward disposition (brokkr-owned; tracked at brokkr R30, "brokkr/worldtree will ping").** The per-turn decay has no room for `decay=f(N)` under preserve-persistence + the A/P-not-1.9 finding → R30's decay is being redesigned as a HYBRID wall+turn decay (brokkr pre-scope). R30 v1 ships GAIN-only (N→negative-reactivity) with decay held at the measured 0.95. My dedicated per-axis A/D run is DEFERRED into the hybrid-decay design pass (one wall-clock-spaced run does per-axis + a turn-vs-wall probe together). Phase-2 (moody-lofn GAIN-direction validation) waits on worldtree's `dynamics_from_ocean()` impl.
|
||||
- `[2026-07-03]` **Relational-arc verify DEFERRED — `relations[]` is Bifrost-provider-only (ADR-0009), confirmed both ways.** The relational-dynamics state (relation_context tie-type / agency / warmth / trust) is NOT on the conversation-API `affect_update` snapshot for base agents (keys: pad/dominant_emotion/emotions_active/baseline_pad/mood_drift only) — only in the provider store; worldtree-dev confirmed by-design per ADR-0009 (emitted over `affect.emit`, deliberately off the SSE). So the Wave-0/1/2 verify needs the bound-provider round-trip (provider running + `--bifrost-plane affect` session), its own focused session. worldtree-dev routed the "expose relations[] to non-provider consumers" observability scope call to Vuong; my rec: keep provider-only (YAGNI — ratatoskr IS a provider, gains nothing; no speculative public surface).
|
||||
|
||||
- `[2026-07-04]` **R30 CLOSED on offline-tests + human face-validity (operator steer, relayed via worldtree-dev).** The deployed gap-injection run was confirmatory-not-measuring (against a deployed system the fade is `exp(-dt/tau_shipped)` by construction -> a fit recovers tau_shipped tautologically; per brokkr's S0 reframe it GRADUATES the interim coefficients, doesn't measure them), and the repo's offline tests already cover the OU formula + BOTH directions (`high_N_fades_slower_than_low_N`, `phenotype_high_n_bigger_negative_excursion`). So no Worldtree build; the interim coefficients graduate validated-as-shipped. My gap-injection harness (read/predict/record; write side stubbed; `predict()` reproduced brokkr's N=0 anchors exactly) is BANKED at `diag/r30-gap-injection-harness` for the parked powered true-tau study. [continues R30 forward-disposition 2026-07-03]
|
||||
- `[2026-07-05]` **Authored-history-write primitive proposed -> accepted as Worldtree #347 (Worldtree owns the engine design; ratatoskr = reference consumer).** SillyTavern first-message generalized to a non-generating ledger-write primitive; can't be done client-side (messages `role` = model-role, not author-role). heid panel pressure-test (3/3 convergence) drove the v1 narrowing (append-only, bounded `effects` enum, drop edit/regenerate). Brief `docs/proposals/authored-message-injection.md` (`c457520`); consumer constraints captured in-brief: hide-existence 404-fallback (`022accf`) + assistant-first provider constraint (`7156b25`). Operator (Vuong) ruled the design-direction call (engine primitive + a real provenance/spoofing security surface). [reference-impl posture: we propose the shape, worldtree-dev owns the contract+impl]
|
||||
- `[2026-07-05]` **#347 v1 wire validated as reference consumer (GREEN).** Adopted positions: distinct sub-resource `POST /sessions/{id}/history` (not `generate:false`), model-invisible provenance (first-message immersion preserved), event-silence for authored seed, `seeded` lifecycle phase, per-session idempotency. Three pre-TDD flags folded into contract rev 1.1: assistant-first provider constraint (Anthropic-family 400s; vLLM/openai_compat OK), content limit is BYTES not chars, 409-active-generation for append-narrator. First-message (create-time, assistant, effects=none) fully served; append-narrator served for the assistant-voice subset (system deferred); debug-seed served for assistant turns (user injection deferred to a future import primitive).
|
||||
- `[2026-07-06]` **Sindra role character-rp -> character (operator).** `character-rp` resolves to a reasoning-tuned RP config (`gen-reasoning` + temp 0.75 + RP `extra_body`); `character` = plain non-reasoning (better for immersive RP). Both non-destructive PATCHes (role is mutable; model is NOT -- server: "PATCH accepts only system_prompt and/or role"). #344 (b19) fixed the role->catalog_id display conflation (the `model` field now shows the ROLE); previously it leaked `gen-reasoning`. Set via raw curl (tier3.py CLI has `--model`, not `--role`).
|
||||
- `[2026-07-06]` **Sindra persona/OCEAN DECLARED -> mood fixed (the full diagnostic converged on a stale personal container).** Root cause of stuck-neutral mood: her OCEAN was prompt-TEXT only, never a structured persona; fix = delete+redefine with the define-time `persona:{ocean:{...}}` field (immutable via PATCH). My diagnosis surfaced a real engine bug **#348** (single-letter vs spelled-out OCEAN keys -> declared OCEAN silently -> 0.0/neutral; worldtree-dev fixed in b21/b22) AND a **stale-container deploy race** (personal's b22 deploy was a pull-only no-op; infra-ops force-swapped run 8211). VERIFIED: bound mood-smoke reads (0.448, 0.267, 0.316) ~= OCEAN-derived setpoint (0.418, 0.249, 0.328). [consumer/provider thesis: "reset + smoke" flushed out two upstream problems]
|
||||
|
||||
- `[2026-07-06]` **OpenAPI re-vendored 2.2.0->2.3.0 (`75da676`, pin-only no bump).** worldtree-dev shipped #347 as spec 2.3.0 (`879cefe`, = the deployed personal b22 image); the SessionStart drift-check flagged our openapi pin STALE. `canonical_sync` pulled 2.3.0; updated the 4 pin-tracking files (`.corviduo-canonicals.toml`, vendored openapi.json, SPEC-PIN.md, pyproject `worldtree-spec-rev`->879cefe). #347 is OpenAPI-only (prose + server contract byte-unchanged, SSE unchanged=event-silent). The re-vendor re-opened the coverage-audit with one new in-scope path-group (the #347 route).
|
||||
- `[2026-07-06]` **#347 authored-history-write CONSUMER SIDE SHIPPED (`v0.19.6`) — direct in-session TDD.** `write_authored_history(client, session_id, *, content, idempotency_key, author="assistant", effects=None, claimed_original_at=None) -> dict` (POST /sessions/{id}/history; body server-pinned `AuthoredWriteRequest` extra="forbid" so omit null effects/claimed_original_at; 200-replay/201-fresh both -> ack dict; **404 -> `AuthoredHistoryUnavailable`** NOT SessionApiFailed = the hide-existence "feature-absent, never probe" contract; 409/422->SessionApiFailed) + `get_session_messages` (un-deferred GET /sessions/{id}/messages, the seed read-back proving model-invisible provenance) + a `--seed-first-message "<c>" --agent <id>` one-shot probe (create session -> seed -> read-back; 404->benign feature-absent exit 0). Contract #2 amended (2 FNs, validated OK) + 19 tests (12 wrapper + 7 cli). Suite **601 green** (clean env; the 2 "fails" under `source env.sh` are the RATATOSKR_ADMIN_API_KEY env-leak into TestParseArgs, not a regression). Coverage: **REST 19/41** (`docs/coverage-map.md` re-converged). Patch bump (coverage tail; consistent w/ the Tier-2 v0.19.1 cadence). **Live-proof pending** the `session.history.write` grant (infra-ops `01KWW3KQEY`). heid-code-review NOT run (offered).
|
||||
|
||||
- `[2026-07-06]` **Tail-2 SHIPPED (`v0.19.7`) — Tier-3 prose docs re-vendored + persona_state body-shape aligned.** worldtree-dev landed the Tier-3 persona/memory/persona_state PROSE docs (`c9e59ec`, on origin) — they serialize as freeform `Any` in the OpenAPI JSON, so the **prose is their source of truth** (my earlier "2.3.0 = #347-only, tail-2 collapsed" was half-wrong: the JSON was #347-only but the prose is separate). Re-vendored `docs/conversation-api-spec.md` (markdown pin, tolerate_drift; `worldtree-spec-rev` 879cefe->c9e59ec, SPEC-PIN history row added). **Consumer fix:** `--set-persona-pad`/`_set_persona_probe` was sending `{pad:[list]}` but the canonical SET body (#317) is `{pad:{pleasure,arousal,dominance}}` (named dict) — aligned it + added a len!=3 guard; updated contract #2 note + set_persona_state docstring + tests. The `set_persona_state` WRAPPER was already correct (freeform pass-through); only the CLI probe drifted. TDD (probe test asserts the dict; +1 wrong-count test). Suite **602 green**, ruff clean. **heid-code-review on #347 (dispatched + returned this session): UNANIMOUS ZERO DRIFT** (Gróa/Hulda/Regin all confirmed the hide-existence 404->`AuthoredHistoryUnavailable` routing holds at wrapper/probe/test layers + the extra="forbid" body-omission + the deliberate write-vs-read 404 asymmetry — confirmation-not-discovery for a well-TDD'd slice against a prescriptive contract). worldtree-dev foot-guns banked in SPEC-PIN + [[reference_worldtree_affect_surface_map]]: ocean single-letter `{O,C,E,A,N}` on /agents/define (#348) vs spelled-out on /characters; memory `{embedder_version, tier3_dreaming}`, stm_* deprecated, allows_world_scope removed->422; only `valence` still 422s.
|
||||
|
||||
- `[2026-07-06]` **Sindra rewritten onto a #347 authored first-message + first-message-preset AUTO-SEED SHIPPED (`v0.19.8`).** Operator "rewrite Sindra" now that #347 first-messages work. Her card had a `**Startup:**` block (a pre-#347 workaround: "introduce yourself + ask for Intensity/Mood/Willingness" with a verbatim scripted greeting) — precisely what #347 replaces. Rewrite, all NON-destructive: **(1)** lifted her scripted opening into a #347 first-message (punctuation-fixed); **(2) PATCHed her live definition** — `PATCH /agents/ratatoskr:sindra` (body `ConsumerAgentPatchRequest` = system_prompt+role, extra=forbid; keeps OCEAN/persona/memory) removing the Startup block -> a 1-line `**Opening:**` fallback + reworded the axes-persist line (25686->25449 chars, verified Startup gone); **(3) codified auto-seed:** NEW module `src/ratatoskr/first_message.py` (`FIRST_MESSAGE_PRESETS` dict {agent_id->text} + `seed_preset_first_message` best-effort helper) wired into ALL 3 session-create paths — cli `_amain` (`--send --new`), tui `_resolve_then_run` (bare `--new`), web `_create_session_endpoint` (POST /api/sessions) — so every new Sindra session opens with her greeting. **Best-effort (INV-001: swallows AuthoredHistoryUnavailable/SessionApiFailed/httpx.HTTPError -> NEVER blocks create)**; per-content idempotency key (`ratatoskr-preset-`+sha256[:12]). Contract `docs/contracts/first_message.contract.md` (module-scoped: `module:`+`purpose:`+`touches:` required, NOT `target_module:`) + TDD (9 unit + 1 web wire-in; **the 3 existing sindra bind tests needed a history-endpoint mock** since creating a preset agent now auto-seeds). Suite **612 green**, ruff+mypy clean. **LIVE-PROVEN generation-free**: create sindra session -> auto-seed -> read-back seq-0 assistant greeting (409 chars). Sindra's greeting now lives canonically in the preset registry (repo); her server card no longer carries it. Patch bump (single-commit feature, no downstream coordination). **FOOT-GUN: sindra requires `end_user_id` on session-create (422 `end_user_id_required`) — all real paths pass it from env (RATATOSKR_END_USER_ID) / web server config.** **Then the full quality gate (operator-directed, folded into v0.19.8): heid-code-review (unanimous ZERO implementation drift; 2 test-only fixups — INV-004 verification-claim made explicit re the global rglob test + an exactly-one-POST assertion) + heid-bug-hunt (3/3 convergence caught what the conformance lens structurally COULDN'T — the code matched the contract's NARROW 3-type ERROR_ROUTING, but INV-001's "NEVER raises" is BROADER). HARDENED: broad `except Exception` → None (re-raise `asyncio.CancelledError`, itself a BaseException), soft-guard PREs (return None, NOT assert — a wiring bug can't crash the create path it's wired into), and `asyncio.wait_for(_SEED_TIMEOUT_S=10s)` bounding the seed write (the CLI/TUI clients run read=None for SSE → a stalled /history would otherwise block create forever). Suite 615 green. LESSON: code-matches-ERROR_ROUTING ≠ honors-broad-INV-001 — heid-code-review confirms contract-conformance, heid-bug-hunt catches robustness gaps the contract's own narrow clauses miss; run both.**
|
||||
|
||||
- `[2026-07-06]` **Web UI now RENDERS the seeded first-message (`v0.19.9`) — operator-reported "i don't see Sindra's greeting on the web ui".** Diagnosis: the auto-seed WORKED (greeting was in the ledger at seq-0), but the web SPA never fetched a session's EXISTING history — NO `/api/sessions/{id}/messages` route (GET /messages was originally deferred out-of-scope; sessions used to start empty so it never mattered) and `startSession()` went straight from create → persona/tools/admin hydration, so the transcript only filled from the live turn stream + user echoes. Fix: (1) NEW web proxy route `GET /api/sessions/{id}/messages` → `get_session_messages` (mirrors the tools/bifrost proxies; status-preserving `session_messages_unavailable` envelope); (2) SPA `loadTranscript(sessionId)` — fetches the route on open, renders assistant items as `.response .md-body` (markdownSafe, same escape-first path as appendResponse) + user items as `.prompt-echo` (textContent), called in `startSession` after the workspace opens; best-effort (swallows failures). Contract `web_debug_surface.contract.md` amended (server endpoint + loadTranscript entries). TDD (2 web route tests, suite 617 green) + **Playwright DOM check PROVED the render** (drove the real UI: pick sindra → open → her greeting bubble appears — the JS-render lens unit tests can't reach; [[feedback_debug_surface_uses_canonical_surface_only]] cousin lesson). Web restarted on the fix. **FOOT-GUN (self-inflicted): `pkill -f "ratatoskr-web --host"` SELF-MATCHES the bash command running it → exit 144, killed its own restart mid-flight — kill the web by PID, never `pkill -f` on a pattern your own command contains.** **FOOT-GUN: uvicorn hangs on SIGTERM with an open admin-events SSE → needed SIGKILL.** **Playwright: python module absent from the venv; use node + `executablePath=/opt/ms-playwright/chromium-1223/chrome-linux64/chrome` — the shared browser is build 1223, npm-latest playwright wants 1228 (version-mismatch), so pin executablePath instead of letting playwright resolve.**
|
||||
|
||||
- `[2026-07-06]` **Web UI: pivot from incremental CSS polish to a designed prototype (Claude Design) that I wire into.** Operator saw an Australis polish pass ("looks fine, but we're attacking it differently") and chose the prototype route — a designer builds the visual shell, I wire real data/SSE into its DOM. Authored the full design brief `docs/design/ratatoskr-web-design-brief.md` (complete information inventory of every pane/datum/state + Australis direction + single-file/no-CDN/vanilla wire-ability constraints). **Tracking surface:** the brief file + Claude Design project `bc0b65d1-a33e-422a-8bc1-3635c9112775` (file `Ratatoskr Console.dc.html`). Import mechanism = the `DesignSync` MCP; blocked on `/design-login` (claude.ai design scopes) — see Current state for the post-auth wiring plan.
|
||||
- `[2026-07-07]` **Affect-egress reference delivered by worldtree-dev + a we-framing render DRIFT-WATCH banked.** worldtree-dev shipped `docs/affect-egress-consumer-reference.md` (`74d2408`, their origin/main) — the authoritative DELIVERED-on-wire vs HIDDEN (system-prompt-only) map for our affect surface. Confirms the v0.20.x console consumes it correctly: DELIVERED via affect.emit = pad + relations[RelationEdge] + dominant_emotion (**type-only, b23**; the #204 `affect_update` SSE is SUPPRESSED for Tier-3, so its richer `emotions_active` is Tier-1-only — we don't get it, and correctly poll our own affect store post-turn instead). HIDDEN render OUTPUTS are deterministically reconstructable from the canon; we reconstruct only the two FULLY-reconstructable (mood descriptor via `canonMood`, relationship directive via `canonDirective`) and SKIP the PARTIAL mood-directive (dominant_emotion is type-only/no-intensity → can't tell if the OCC directive fires at salience≥0.2 vs the PAD-band fallback) — honest per INV-001. Vendoring the ref doc as a `tolerate_drift` pin is SURFACED to Vuong (worldtree-dev will co-sign + honor a ping-on-change handshake, same as the d2-canon). **WE-FRAMING DRIFT-WATCH → STOOD DOWN (worldtree-dev 2026-07-07, `01KWXMQPHN…`).** The b24 3-gate we-framing conditional (drop "; avoid premature we-framing" under unsafe_capable+deep-warmth+expressive) was **REVERTED** — Vuong reframed it. So **`canonDirective` STAYS UNCONDITIONAL** (always appends the clause), which is CORRECT: it matches the currently-deployed renderer (b23) AND my pinned `affect-egress-consumer-reference.md` §2d (the doc reverted too — nothing changed on the wire or in my pin; NO re-vendor needed). The HOLD was right → ZERO rework. **NEW forward-watch (replaces this one):** the we-framing gate was a SYMPTOM — the render bakes enterprise safety-guards into the directive strings, so even `full`-tier characters get DEFANGED emotions (a hostile villain told to "keep a firm emotional boundary"). Fix = a **use-case-segregated persona render** (assistant / companion / RP-gaming), now a fresh **brokkr deep-research epic**. When it lands + is implemented, the render behavior for character/RP agents (→ our `canonDirective` + `canonEmotionDirective` reconstruction + the vendored d2 canons) will change MATERIALLY — worldtree-dev re-engages ratatoskr-dev then with the new reconstruction spec + a coordinated re-vendor. Until then: NO action, reconstruction stays as-is. [my unsafe_capable/mood_tier heuristic — character→full, agent→safe — was confirmed correct + is banked for whenever tier-gating returns]
|
||||
- `[2026-07-12]` **R34/R35 character-self-report reframe LIVE (WT b53); affect-half VALIDATED in prod, powered memory-half eval GREENLIT + designed.** worldtree-dev's reframe: affect + memory now driven by the character's OWN model self-report on our RP seat (Deckard/Magidonia), replacing external Vili inference; conditioned on an authored `persona.psychological_profile` (else a deterministic OCEAN scaffold). Affect-half smoke GREEN in prod (bound sindra turn on Deckard → contextually-apt `disappointment`); TIER LOCKED (Tier-3 bound-character path). Vuong approved brokkr's prereg for the powered eval (Q1 salience-divergence / Q2 floor-recall / Q3 firewall / Q4 graded-slot / Q5 authored-vs-scaffold / Q6 sliding; ~300 drive-runs). Role split: brokkr generates sets + authors ground-truth + scores; ratatoskr DRIVES the b53 producers; worldtree supplies the pair + a producer-probe. Three eval characters authored + peer-validated (sindra relational + Torvald low-A operational = divergence pair; Ilva high-N = affect-magnitude arm) — wire-ready in `scratchpad/eval_profiles_WIRE_READY.md`. Full in-flight detail in Current state; eval thread althing `01KXAN073B`.
|
||||
- `[2026-07-12]` **Capture-path resolved → a worldtree producer-probe (ratatoskr-caught blocker).** The R34/R35 memory extraction runs ONLY at promotion (idle-10min / session-close / turn≥6), never per-turn — so single-turn eval exchanges have no store to read, and reading the promoted store would confound producer-efficacy with promotion-policy. Fix (worldtree, pending Vuong's greenlight on a new gated-eval endpoint): a dedicated producer-probe returning raw {notes,facts,floor} pre-promotion, decoupled from session state. Affect self-report stays on the proven bound-turn → `:8392` /affect/state path. Also caught + confirmed this arc: (i) the Q1 system_prompt confound → neutral-for-all prompts so divergence is attributable to the persona layer, not the base prompt; (ii) the OCEAN-scale question → storage `[-1,1]` (soong-dev-confirmed), and the b53 producer maps `(v+1)/2 → [0,1]` before the disposition bands, so `[-1,1]` renders correctly = a NON-issue (no mis-render). Standard define (persona + memory:{}), no per-agent promotion config.
|
||||
- `[2026-07-12]` **bifrost 1.1.1 adopted (v0.20.10, pushed) — the library-level fix for the frozen-v0.6 handshake leak.** 1.1.0's `describe_store` leaked the v0.7-additive `sortable_chunk_fields` into a v0.6-negotiated StoreCapabilities → a strict v0.6 client rejects our `:8392` handshake; 1.1.1 gates additive fields on the negotiated wire (ADR-0008). Zero-code patch repin (provider extra 1.1.0→1.1.1 + uv lock), 631 green (clean env — the 2 `test_cli` failures were an env leak from `source env.sh` exporting `RATATOSKR_ADMIN_API_KEY`), committed `62a16d2`, provider restarted on 1.1.1 (PID 3242269). Not exercisable by our v0.7 WT peer (the v0.6 fix needs a v0.6 client), but the running server is now v0.6-clean. Reference-impl adopt-canonical (bifrost-dev flagged). Also: `contract-drift-check-v1` canonical single-pin-synced (`7bca76e`, pushed; `canonical_sync.py` has no single-pin flag so did it by hand — the 2 `tolerate_drift` worldtree prose pins are deliberately held STALE for the R34/R35 eval diff-review).
|
||||
- `[2026-07-10]` **Bifrost bound-handshake blocker → bifrost 1.1.0 broke frozen-v0.6; fixed by Worldtree b47/wire-v0.7, our side needed ZERO change.** Surfaced while running the R34-v1 affect.emit live-verify: a bound Tier-3 session-create to our `:8392` provider began failing `bifrost.schema_validation_failed` (had worked 2026-07-07, pre-b35). ROOT CAUSE (worldtree-dev-confirmed, path-ii/response-side): bifrost 1.1.0's `describe_store` EMITS `sortable_chunk_fields` (the v0.7-additive field) REGARDLESS of the negotiated wire → a v0.6-negotiated handshake RESPONSE carries a v0.7 field → a v0.6 peer's `additionalProperties:false` rejects it. Bit because BOTH Worldtree (bumped bifrost 0.9→1.1 in b35 via their #349, client wire-v0.6) AND ratatoskr (my adopt) were on 1.1.0. NOT our bifrost version specifically (failed identically on 1.0.0 AND 1.1.0). FIX (worldtree-dev, coordinated-v0.7 move): WT advanced its client `_WIRE_SCHEMA_VERSION` v0.6→v0.7 (b47 / `4eb374c`), where sortable_chunk_fields is accepted — **our provider (already 1.1.0/v0.7) needed ZERO change; keeping 1.1.0 was LOAD-BEARING** (reverting to 1.0.0/v0.6 would've been INcompatible with b47). **R34-v1 affect.emit verify GREEN on b47:** bound sindra turn (`model=character-rp`/Deckard, 13.1s) → fresh affect.emit `dominant_emotion='disappointment'` NON-NULL with our `affect.full` grant (verified-active on live traffic; ungranted principals get null — the leak-fix). Reported to worldtree-dev + infra-ops. Personal deploy train this session: b35 (R32 render) → b44 (RP seats) → b46 (R34-v1 affect-gov) → b47 (wire-v0.7). **Sindra on Deckard live-confirmed** (reasoning latency ~13s).
|
||||
- `[2026-07-09]` **Sindra PATCHed to `character-rp` → Deckard reasoning-RP seat (operator-directed, INTERIM).** worldtree-dev heads-up (v1.0.0b44, batched, not-live-yet): character-RP roles re-point to dedicated seats — `character`→Magidonia-24B (non-reasoning RP tune), `character-rp`→Deckard-PKD Qwen3.5-27B (reasoning-on). Sindra was on `character` (→Magidonia); operator chose Deckard for her complex stateful mechanics (Intensity/Mood/Temperature/Willingness axes, form-assumption, failure/resurfacing) — reasoning tracks multi-axis state better. Role is MUTABLE via PATCH (non-destructive: persona + memory preserved, prompt unchanged @25449 chars; NOT a DELETE+redefine). She's on the reasoning-RP config now, resolves onto Deckard when b44 deploys. **INTERIM: "until we get a GM type on-board"** — operator plans a game-master-type agent; sindra→Deckard is the stopgap for state-consistency until then, likely revert/rebalance when the GM lands. Wire/API transparent (call the ROLE not the model; model field now reads `character-rp`); old character-rp temp-0.75 override retired → seat's canonical RP samplers server-side.
|
||||
- `[2026-07-08]` **PAD display relaxed for R32-1B unbounded-z, done PROACTIVELY (`v0.20.9`, patch, operator-directed "sindra full and unbounded").** Confirmed (grep-verified, airtight) the PAD clamp is PURELY debug-surface: the only clamps (`_clamp1`/`clamp01`) lived in `web/static/index.html` display fns; the affect store is conduit-opaque, the read route + proxy pass verbatim, and the sole write path (`--set-persona-pad`→persona_state) is UNCLAMPED — ratatoskr is a downstream OBSERVER, so the clamp has ZERO agent-experience/efficacy consequence. Replaced the hard [-1,1] clamp with AUTO-SCALE to the session's own max |PAD| (`padScale` floor 1.0 → `padFillFrac` faders + `_padNorm` orbit): unbounded z renders at full range, never pegs/escapes the frame; today's [-1,1] values unchanged (scale==1); exact value always shown numerically. Playwright-verified (z=±6.2 → faders ≤50%, orbit in-box, +6.20 readout, zero regression at scale 1). When worldtree-dev pings R32-1B-shipped, our side already handles it. Sindra's ACTUAL "full/unbounded" affect is Worldtree-side (she's on the default/uncompressed render, NOT the R32 `assistant` compressor) — nothing for ratatoskr to change there.
|
||||
- `[2026-07-07]` **Web-UI iteration-3 SHIPPED (`v0.20.7`, patch) — all three queued items + the memory 0/0 root-caused.** (A) design iteration-3 (sparkline grid-bg, PAD Δ-bar strips replacing the polyline strips, dimetric-open-box mood orbit w/ JS replay) adapted into index.html; (B) memory viewer (provider `GET /memory/chunks` debug read on `:8392` → web `/api/memory/chunks` proxy → MEMORY console pane) + the mystery SETTLED: bind grants memory fine (`caps_granted:[memory,affect]`) but sindra emits ZERO memory ops → her reset-clean def lacks `memory:{}`; PROVEN via throwaway `ratatoskr:memprobe` (memory-enabled) → 4 real chunks landed + rendered in the pane; (C) markdown pass-2 (tables/nested-lists/ordered-start/streaming). 631 tests green; `:8392`+`:8765` restarted on new code. New env `RATATOSKR_MEMORY_READ_URL=:8392`. Contract `web_debug_surface.contract.md` amended in-commit. **Leftover:** memprobe agent + its 4 test chunks live in `memory.db` (optional cleanup). See Current state § ✅ SHIPPED for full detail. [supersedes the QUEUED entry below]
|
||||
- `[2026-07-07]` **Web-UI next-work QUEUED for a fresh-context session (operator-directed `/snapshot` handoff) — full specs in Current state § ⏭ QUEUED.** Three deferred-but-scoped items: **(A)** Claude Design prototype ITERATION-3 re-import (sparkline bg grid `<pattern>`, PAD strips → per-turn-Δ-bar HTML columns, mood orbit → DIMETRIC open-box az35/el25 D-right/A-left-back/P-up + JS-animated replay `orbitDynamics`); **(B)** memory viewer console pane (provider `GET /memory/chunks` read → web proxy → polling pane, mirror #18-D2; bundle a 6-turn bound round-trip proof to settle the 0/0-memory mystery); **(C)** markdown pass-2 (tables / nested lists / streaming). **Tracking surface:** the Claude Design projectId `bc0b65d1-a33e-422a-8bc1-3635c9112775` (durable — re-pull for exact coords) + this snapshot capture; operator-directed. Design scopes already granted (no `/design-login`).
|
||||
- `[2026-07-07]` **Markdown pass-1 SHIPPED (`v0.20.6`) — RP semantic coloring + paragraph reflow (operator-directed markdown rework, step ①).** The transcript renderer (`markdownSafe`/`mdInline`) now colors the two roleplay registers: `"quoted"` dialogue → SPEECH (bright `--md-speech`=fg-0), `*asterisk*` → ACTION/narration (muted-italic `--md-action`=fg-3, on `em.md-action`); both tunable via 2 CSS vars that cascade through `--fg-*` so they auto-adapt to the light theme. Plain text stays default narration. **KEY ORDERING:** speech-wrap runs BEFORE the em/link passes so a generated `class="md-action"` / `href="…"` quote can't be mis-read as dialogue (adversarially verified). Straight + smart quotes; apostrophes don't trigger; unbalanced/half-streamed quotes stay uncolored until they close. ALSO fixed the ugliest existing bug: single newlines were hard-`<br>`s → now CommonMark soft-breaks (space); a hard break needs 2+ trailing spaces or a trailing `\`. INV-004 escape-first preserved (html inside a quote escaped). Verified: 10-case Playwright unit-check of `markdownSafe` (speech/action/attr-trap/apostrophes/mixed/reflow/hard-break/escaping/unbalanced/paragraphs) all green + a visual render. Patch bump. **Markdown rework remaining: pass-2 (tables / nested lists / ordered-list numbering / streaming robustness); + the MEMORY VIEWER (step ②, console pane recommended) still queued.**
|
||||
- `[2026-07-07]` **Affect-derived tooltips (`v0.20.5`) + transparent squirrel vendored + v0.20.0–.5 PUSHED to origin (operator-authorized).** Native `title` hints on all 6 affect-derived cells (samples/updated/baseline P·A/drift Δv/volatility — each with meaning + Tier-1-vs-Tier-3 availability). Committed the transparent full-res brand mark at `docs/design/ratatoskr-mark.png` (1024², alpha; the bg-removed source the inlined favicon derives from — reproducible via the documented ImageMagick corner floodfill). The web-UI redesign arc (v0.20.0 Claude Design console → .1 sparkline-overflow+tooltips → .2 context-injection → .3 squirrel brand/favicon → .4 SVG sparklines+3D cube → .5 derived tooltips) is now all on `origin/main`.
|
||||
- `[2026-07-07]` **SVG sparklines + 3D mood cube imported from the updated Claude Design prototype (`v0.20.4`) — operator: "the svg sparklines and the new 3d graph".** Re-pulled `Ratatoskr Console.dc.html` via `DesignSync get_file` (designer iterated the same project, +6KB). Adapted 3 SVG systems out of the `.dc.html` into vanilla, replacing the unicode-char sparklines: **(1)** per-PAD-fader VERTICAL strips (`stripPoints`, 26×132 SVG beside each bar — time down Y newest-at-bottom, value on X ±11, `stripFade{P,A,D}` gradient, dot at newest; this also nails the earlier "next to each meter" ask); **(2)** relation-row HORIZONTAL sparklines (`sparkPointsH`, 56×13, auto-scaled, `sparkFade` gradient + end dot — also kills the old unicode-overflow "n behind graph" for good since it's a fixed-width SVG); **(3)** the mood-orbit reworked from a 2D P×A scatter into a **3D ISOMETRIC P×A×D cube** (`proj3`: P right-down/A left-down/D up, 2:1 iso, center 62,66, scale 26 — REVERSE-DERIVED from the design's placeholder now-point + verified: `x=62+26P−26A, y=66+13P+13A−26D`), with the trajectory + pulsing now-marker + a drop line to the D=−1 floor + a floor-shadow ellipse for depth. Gradients in one hidden `<defs>` svg. Removed the orphaned `sparkline()`/`_SPARK`. Contract amended. Playwright-verified (injected 24-sample history: 3 strips + 4 relation sparklines + the 3D cube trail/drop/floor all render; gradients resolve). Patch bump.
|
||||
- `[2026-07-07]` **Brand mark + favicon → the aurora squirrel (`v0.20.3`), replacing the `ᛯ` rune.** Operator supplied `/home/lkraven/rata.png` (chibi cyan-green aurora squirrel + acorn). Removed the black background via ImageMagick corner flood-fill (`-fuzz 20% -floodfill` from all 4 corners — keeps the squirrel's interior black linework/eyes (not edge-connected) + the glow, drops only the connected background), downscaled 1024→80px + quantized-64-colors (~12KB base64), inlined as ONE `SQUIRREL` data-URI const in the JS wiring the favicon `<link id="favicon">` href + both `.brand-mark` imgs (rail brand-row + setup-card h1). `.brand-glyph` (font-rune) CSS replaced by `.brand-mark` (img, drop-shadow glow + breathe). Playwright-verified (both marks + favicon decode, naturalWidth>0; no brand-glyph left). Source PNG stays at `/home/lkraven/rata.png` (not committed — data-URI is self-contained + reproducible via the documented floodfill). Patch bump.
|
||||
- `[2026-07-07]` **Context-injection view SHIPPED (`v0.20.2`) — the console now reconstructs the FULL hidden affect block Worldtree injects into the agent's system prompt (operator: "use that canon in the interface, see as much context injection as possible").** No new canon vendored — the strings were ALREADY in the pinned `d2-mood-render-canon-v1.json`; extended `build_persona_canon.py` to emit `mood_directive {occ_directives(15), pad_band_fallback, salience 0.2, pad_band_cutoff 0.3, full_only[love,anger,disgust,shame]}` into `persona_render_canon.json` (regen via Worldtree venv). New JS `canonPadFallback(pad)` + `canonEmotionDirective(type)` — BYTE-EXACT mirrors of Worldtree `core/persona/renderer._pad_band_fallback` + `derive_directive`; `renderDirective` expanded into a "CONTEXT INJECTION · reconstructed · hidden from consumers" panel showing mood descriptor [exact] + mood directive [candidate] + relationship directive [exact]. **HONEST-PARTIAL (affect-egress-ref §3):** affect.emit is type-only (no intensity) → can't evaluate the salience gate (≥0.2) → show BOTH candidates (OCC emotion directive + PAD-band fallback) with the "injected if intensity ≥ 0.2" caveat, never assert which fires; when dominant_emotion absent the fallback alone is exact. Panel labeled dev-only per the reference's "not-for-end-user-display" caveat (ratatoskr = the sanctioned reconstruct-platform-behavior use). Vendored + pinned `affect-egress-consumer-reference.md` (tolerate_drift, worldtree-dev co-signs + pings on change; drift 6/6 green). Contract amended. Playwright-verified (sindra: dominant_emotion=joy → joy OCC directive candidate + PAD-band fallback both render, exact/candidate tags color-coded). Patch bump (single-commit feature, no downstream coordination; minor-defensible but tie-breaks to patch). **OPEN — SURFACED to Vuong:** take worldtree-dev's standing offer to add emotion INTENSITY to affect.emit → resolves the OCC-directive-vs-fallback EXACTLY (drops the candidate ambiguity). [reference-impl privileged view: ratatoskr shows what WT hides from regular consumers]
|
||||
- `[2026-07-06]` **Claude Design console SHIPPED (`v0.20.0` MINOR, operator-approved) — see Current state for the full record.** Pulled via `DesignSync get_file` (scopes already granted), adapted `.dc.html`→vanilla single-file, wired all `/api/*`+SSE into the new 3-column console DOM, then a round-2 fixup (light theme, full Bifrost pane, ticker-spine fix, per-fader PAD Δ, inlined favicon). 84 web tests + node-Playwright-vs-personal-:8081 both green; contract amended in-commit; INV-001 honest-shape held (canonical mood word for Tier-3, no fabricated emotion). **Foot-guns reconfirmed:** the `.dc.html` dialect is NOT runnable (translate, don't paste); a scroll-container-anchored `::before` timeline spine scrolls out of view on auto-scroll (anchor it to a content-height inner wrapper instead); a favicon 404 shows as a browser `console.error` even when handled (don't count it as a JS-test failure). **Foot-gun (favicon):** operator PNGs are full-res (1024² / 805KB) — downscale to ≤64px before inlining as a data URI.
|
||||
|
||||
- `[2026-07-13]` **P06 memory-half DRIVEN + SCORED — the reframe's memory mechanism is validated.** ratatoskr drove 308 runs clean (0 errors, binding 154/154 both arms), dropped to brokkr, brokkr scored (R35.45). The authored `psychological_profile` causes the memory-salience divergence (authored 0.618 vs stripped 0.235 ≈ null; Δ+0.382) — the effect is the profile, NOT OCEAN leaking (negative control holds). Real effect, below the 0.70 strength bar → optimization phase next, not a re-litigation. See Current state for the full record. Drive role closed both sides.
|
||||
- `[2026-07-13]` **Memory extraction turned REASONING-OFF for the eval (operator-directed).** "same model, reasoning off via explicit kwarg" — scoped to the memory extractor ONLY (affect + RP stay reasoning-ON). The real lever was `chat_template_kwargs.enable_thinking:false` (the naive `thinking_enabled=False` kwarg was a no-op — see Tried/abandoned). Deckard extraction went 45s→~5s.
|
||||
- `[2026-07-13]` **Vendored the brokkr R34 psych-profile canon (Vuong-directed) — BOTH files, not just the parameters.** brokkr said "vendor alongside the authoring-spec you already hold"; I held its content but never a pinned repo copy, so I vendored both (`psych-profile-parameters.md` + `psych-profile-authoring-spec.md`) under `docs/vendor/brokkr-r34-psych-profile/` — makes the parameters' "authoring-spec governs on conflict" clause resolve against an in-tree file, not a dangling pointer. brokkr confirmed keeping both is the better setup. tolerate_drift; brokkr owns + pings on change.
|
||||
- `[2026-07-13]` **Affect-egress "coordinated re-vendor" open item RESOLVED — it was a benign R32-1B doc note, not the we-framing conditional.** The two stale `tolerate_drift` WARN pins re-synced to a 2-line PAD-range note (unbounded-z, already adopted v0.20.9). Re-synced autonomously (zero behavioral impact); the actual we-framing-conditional re-vendor remains future.
|
||||
- `[2026-07-13]` **WT #355 root-caused via ratatoskr telemetry (Vuong-routed via soong-dev).** The char-rp-reasoning turn-never-terminates wedge: over-budget `trim_messages` return (last-2 msgs + system + 8 bifrost tool schemas > input_budget = context_window×0.7) triggers the seat hang; worldtree confirmed + found the terminal-suppression (300s stall-watchdog cancel stuck in httpx `AsyncShieldCancellation`). Fix landing WT-side (Slice-C cancel-independent terminal). Two-proof localization (consumer-clean + tool-less-clean → WT-side tool-loop). Standing loop-in obligation on soong's next re-trigger.
|
||||
- `[2026-07-13]` **WT #355 VALIDATION — fix CONFIRMED; the standing loop-in obligation is DISCHARGED.** The fully-instrumented re-drive ran, coordinated from the ratatoskr seat: infra-ops armed a full WT-netns pcap + py-spy (T0/30/60/300) on the b60 :8081 container; soong drove an 8-turn accumulating RP-with-tools repro on a FRESH session. Authoritative WT-side turns-table: wedging turns 2064/2065 → `completed=True, cancelled=1, phase=STALLED`, dur 302s/360s — the 300s stall-watchdog + Slice-C cancel-INDEPENDENT terminal fired cleanly, vs the pre-b60 baseline (turn 2061) 16-min hang / NO terminal. Slice-B `_log_wedged_task_stack` named the frame (`agent_turn.py:586 async for chunk in stream_iter`, idle-in-epoll — the wedge was a thinking-phase over-budget hang, NOT the attach_tool precursor first assumed). soong's client verdict is 45s-masked (soong-lab v0.3.2 idle-timeout) → NOT b60's terminal; the WT-side capture is authoritative. Threads `01KXE0MXDX…`(wt) / `01KXE0X2DD…`(infra) / `01KXE0X6GR…`(soong).
|
||||
- `[2026-07-13]` **b61 adopted as the personal target — the orthogonal over-budget TRIGGER also fixed.** worldtree shipped b61: the provider stream loop terminates on `finish_reason` + a per-read idle deadline + a 300s wall-clock backstop (no longer waits on the SDK `[DONE]` sentinel), plus a custom llama.cpp reasoning-budget multi-terminator seat → the runaway is bounded at BOTH layers. The #355 STICK (no-terminal) and its trigger (why it wedges) are now separately fixed. Resume-durability gap → **WT #356** (worldtree-owned).
|
||||
- `[2026-07-13]` **Cleaned + prepped for a Sindra run (operator: "clean up everything + prep").** Reset provider stores to 0/0 (`reset-sindra-stores.sh`, rolling backup `db-reset-backup/`); deleted the throwaway `ratatoskr:memprobe` agent via the operator's `!` (destructive DELETE trips the auto-guard — reverses the earlier KEEP). `ratatoskr:sindra` verified present + persona-intact on b61. Environment Sindra-run-ready (web :8765 + provider :8392 both up, single healthy provider instance); operator driving the run interactively.
|
||||
- `[2026-07-14]` **soong-lab adopted as our Tier-3 agent-authoring studio (operator-directed).** ratatoskr consumes soong-lab bundles → WT `agents.define`, and ships agents back as bundles. Sindra round-trip proven (soong imported her `resume` half through real `import_bundle`); soong-lab export+importer contracts pinned via canonical-sync (`canonical_source=soong-lab` @ f434016, commit `39050c3`). The 4 soong-lab ROLE_CHOICES = WT model-role slugs 1:1 by name (worldtree-dev); `character`/`thoughtful-character` need the `character` grant (held), `assistant`/`thoughtful-assistant` need `foundational` (routed to infra-ops). NOT on the v1 coverage-map (sibling-studio interop, not a WT I/O point) — operator chose to pursue anyway.
|
||||
- `[2026-07-15]` **Cross-session recall failure root-caused → the person-prime `scan` build.** worldtree-dev: WT injects a recalled fact only if combined score (sim×salience) ≥ 0.45 (`auto_inject_combined_score_threshold`) AND recall is per-turn query-gated → moderate-sim durable facts (name ~0.40) never inject. Designed turn-0 fix = WT #349 person-prime (query-less top-N-by-recency injection), capability-gated on the store advertising `updated_at` sort — dark for our provider. Fix = implement the sorted `scan` verb + advertise `sortable_chunk_fields` (ZERO WT change). Contract-first (un-defers bifrost `scan`); skipped heid-contract-review (external spec from worldtree-dev, validated point-for-point). **SHIPPED + deployed + live-verified v0.20.14** → `persistent-memory.d/2026-07-16-person-prime-scan-shipped.md`.
|
||||
- `[2026-07-15]` **Sindra redefined from the soong-lab bundle** (`/tmp/sindra.json`, operator "update to match"). Persona immutable → DELETE+redefine; `memory:{}` preserved; motivational string→WT-object mapped (synthesized id/type/salience, flagged to operator); role `character-rp`→`thoughtful-character` (same seat). Payload validated via a throwaway probe (`sindra-probe2`→201) BEFORE the destructive delete. first_message preset updated + web restarted.
|
||||
|
||||
- `[2026-07-16]` **WT #364 + brokkr R39 re-drive (DECISIVE) — name-recall root-caused Worldtree-side; the fix is the signal FAMILY, not a threshold.** No (sim,salience) fusion can fix it (stale negative Pareto-dominates the true name); our specimen + operator's subject-provenance catch (Sindra's own prompt-behavior leaked into user memory) shaped #364's `(subject,relation)` slot-supersession + identity-tier fix. → `persistent-memory.d/2026-07-16-wt364-r39-name-recall.md`
|
||||
- `[2026-07-16]` **bifrost ruled scan snapshot-cursor NORMATIVE (offset NOT blessed) + shipped conformance coverage in bifrost 1.1.3.** Our multi-page offset cursor is now known-non-conformant (single-page person-prime is fine, nothing shipped is broken); adoption is operator-sequenced. → `persistent-memory.d/2026-07-16-bifrost-cursor-conformance.md`
|
||||
|
||||
- `[2026-07-16]` **brokkr flagged ratatoskr as R39 Phase-2 PROBE-RUNNER (Arm-2 contradiction set = our frozen 10-chunk specimen); ACCEPTED IN PRINCIPLE (Vuong 2026-07-16) — formal scope + effort commit lands when Arm-2 actually spins.** Non-urgent — downstream of worldtree's #364 impl; brokkr pings the export shape when Arm-2 spins. Gate: reproduce AUROC~0.59 (similarity can't separate contradiction from paraphrase) on the domain set BEFORE crediting deterministic slot-supersession; eval is a SURFACING test not retention (injection=0 hard-gate + surfacing-recall≥0.95, §5.5). Prereg: brokkr `research/R39-memory-salience-dreams-surfacing/phase-2/preregistration.md` (R39.9). Tracked at brokkr's Phase-2 prereg + the Arm-2-spin ping (thread `01KXMT42Z7…`); I sent a non-committal receipt (role routed to operator).
|
||||
|
||||
_41 older entries (2026-05-* — the original debug-TUI/web build era) archived to archival-memory.md._
|
||||
|
||||
_For per-issue TDD implementation notes, Volva findings, and contract amendments, see the git log — every per-issue commit carries a structured message capturing the trail._
|
||||
@@ -174,6 +264,7 @@ _For per-issue TDD implementation notes, Volva findings, and contract amendments
|
||||
Log of approaches that were tried and rejected, with rationale. Future-self
|
||||
defense against re-attempting the same cul-de-sac.
|
||||
|
||||
- `[2026-07-10]` **Reverting our provider to bifrost 1.0.0 to fix the bound-handshake `schema_validation_failed` — DISPROVEN, and it would've been the WRONG state.** Hypothesis: "my 1.1.0 bump broke the handshake; revert fixes it." Reverted 1.1.0→1.0.0, restarted `:8392`, retested → STILL failed (1.0.0 fails too). Root cause was Worldtree-side (their bifrost 0.9→1.1 bump in b35 broke frozen-v0.6), fixed by WT adopting wire-v0.7 (b47); **staying on 1.1.0/v0.7 was the RIGHT state** (1.0.0/v0.6 would've been incompatible with b47). **Lessons: (1) don't bump a WIRE-PROTOCOL dependency out-of-lockstep with the peer on the other end of the wire; (2) a library's "additive/non-breaking over frozen vN" claim can fail if it emits new fields regardless of the negotiated version — bifrost 1.1.0 emitted `sortable_chunk_fields` on a v0.6 handshake; (3) diagnosis foot-gun: our provider logged handshake 200 (WE accepted) but WT rejected our RESPONSE, so the error surfaces at session-create direction-ambiguous — op-feed the handshake req/resp to disambiguate our-provider-rejects-request (i) vs peer-rejects-our-response (ii).** Also: this sandbox BLOCKS foreground `sleep` (SIGTERMs the command, exit 144) — use separate tool calls / Monitor-until-loop for waits, never `sleep` in a compound command.**
|
||||
- `[2026-06-15]` **"Sindra hasn't been registered" was an under-verified inference — WRONG.** Concluded it from grepping ratatoskr's CODE (`sindra` absent from `src/`), but Tier-3 registration is SERVER-SIDE (`POST /agents/define`) — a code grep structurally can't see it. **Rule: to check whether a Tier-3 agent exists, query the Worldtree instance, never the consumer repo's code.** (Extended 2026-06-17: even `GET /agents` can't see consumer agents; only `GET /agents/<owner>:<name>` with the owner key does.)
|
||||
- `[2026-06-14]` **Artifact-only contract review can't validate against a dependency's ACTUAL behavior.** `/heid-contract-review` sees only the contract, never the external library (bifrost) — so "the consumer under-built against bifrost's real semantics" is invisible to it by construction (the affect idempotency model shipped wrong because of this). Real-lib TDD against the shipped library + the executable reference store + the #195 parity test are the gate. Don't treat a clean contract review as evidence the code matches the dependency.
|
||||
- `[2026-06-15]` **"byte-equal" round-trip slip propagated affect→memory via copy-paste.** The affect contract's byte-identical→semantic fix reappeared in the memory contract's INV-001 (sibling copy). Only an INDEPENDENT `/heid-contract-review` of the memory contract re-caught it. **Paraphrase every sibling contract fresh — don't amortize one review across a family; copies carry the parent's slips.** (also a feedback auto-memory)
|
||||
@@ -201,5 +292,27 @@ defense against re-attempting the same cul-de-sac.
|
||||
- `[2026-06-20]` **The post-turn-async timing trap bit AGAIN — even a 35s post-`[done]` read missed the promotion `upsert_many` by ~2s** (it landed `19:48:58`; the read was ~`19:48:56`). A 15s-interval background poll caught it on the first tick. Same family as the affect.emit / async-promotion traps already logged — re-confirmed that "wait once then read" is fragile for post-turn writes; **poll a window, don't snapshot once.** (The affect.emit write, by contrast, DID land inside the 35s window — promotion is the slower of the two post-turn writes.)
|
||||
- `[2026-06-30]` **Heimdall keys are PER-INSTANCE — a key minted on one Worldtree 401s on another.** Our Conversation-API key works on personal `:8081` but 401s `auth_invalid` on demo `:8080` (per-instance Heimdall user store + pepper; fresh deploys start with an EMPTY key store). Same as the admin key (personal-only). **To live-drive a given instance you need a key minted FOR that instance** (request via infra-ops). Couldn't live-prove the b2 409 on demo for this reason → deferred to personal-b2 where we have access.
|
||||
- `[2026-06-30]` **`tea comment <N>` hangs on Gitea** (the whole compound bash auto-backgrounded + stuck on the open `tea` call). The #11 prereq comment hung; killed it + posted via the Gitea HTTP API directly (`POST /api/v1/repos/vh/ratatoskr/issues/<N>/comments`, token from `~/.config/tea/config.yml`). **For issue comments, prefer the Gitea API over `tea comment` when `tea` is flaky** (CLAUDE.md already says use HTTP for comment-EDITS; this extends it to ADD when tea hangs). Verify-then-post (check the comment didn't already land) to avoid a double-post after a kill.
|
||||
- `[2026-07-02]` **Mask-HOSTED transient characters have a STATIC mood engine — cost a whole R29 probe.** A first probe used a `POST /characters` transient character bound via `agent_id=mask` + `character_id`; its PAD sat at baseline across 15 praise/contempt/dominance turns — the appraisal→PAD engine does NOT run on the mask-hosted transient-character path. The dynamics run only on BASE persona agents or a session bound to ratatoskr's affect provider. **To probe mood dynamics, use a base persona agent, never a mask-hosted transient character.** (mask AS a base agent — `agent_id=mask`, NO `character_id` — DOES run the engine, neutral 0,0,0 baseline.) [auto-memory `reference-worldtree-affect-surface-map`]
|
||||
- `[2026-07-03]` **The "neutral non-appraising tail" premise fails — the neutral MESSAGE choice dominates.** The R30 φ0 method assumed neutral turns don't re-appraise, but factual-question neutrals ("capital of France?") trigger a new emotion nearly every turn (disappointment from the warmth-withdrawal let-down after a positive impulse) → `emotions_active` never empties in 50 turns. A minimal "Please continue." triggers FAR fewer (emotions clear ~turn 16 with spacing). The personal dry-run caught this BEFORE ~280 demo turns were spent on it — the instrument catching a flaw in the measurement design before the compute burn. (Irrelevant to the joint fit — the push_t covariate handles re-appraisal — but load-bearing for the empty-tail read.)
|
||||
- `[2026-07-03]` **Two φ0-fit traps: fast-turn timescale + low-baseline conditioning.** (1) At fast turn cadence the per-turn PAD decay (φ≈0.95/turn) reaches the anchor LONG before the ~200s wall-clock emotion fade → no signal in the (eventual) emotion-free tail; need wall-clock SPACING (~16s) so the fade lands while PAD still has signal. (2) A low-baseline agent's impulse in the constrained direction (forseti P0.239 negative) gives a tiny excursion → ill-conditioned regression (r²=0.46) that FALSELY tripped "config≠behavior" when its φ was averaged in. **Weight/exclude by fit quality (r²) before aggregating — a signal-poor run isn't evidence against the config.**
|
||||
|
||||
- `[2026-07-06]` **`persona_state` + the agent envelope are Tier-3-BLIND -- NOT valid signals for "did a persona store".** `GET /agents/{id}/persona_state` returns 404 `persona_not_configured` for EVERY Tier-3 colon-id (hardcoded short-circuit, `api.py:1266` "regardless of row state"); the `ConsumerAgentResponse` envelope never echoes persona/motivational/memory (`api.py:538`). I mis-called "persona didn't store" from these two blind reads -- the **201-not-422 on define IS the store-success signal.** To actually SEE a Tier-3 mood, read the emitted PAD off the Bifrost affect egress after a BOUND turn (Tier-3 persists nothing Worldtree-side per ADR-0009; no persona/mood READ endpoint).
|
||||
- `[2026-07-06]` **Raw `POST /sessions` is NOT Bifrost-bound -> zero affect/memory emits.** The web surface binds by setting the `bifrost` block on session-create; a raw session doesn't -> 0 affect rows, which I nearly misread as "mood is neutral". Bind from the CLI with `--new --bifrost-url http://10.100.10.50:8392` (the combined provider). Gotchas: `--bifrost-plane affect/memory` map to the SEPARATE `:8390`/`:8391` providers (`endpoint_for_plane`), which I'd PRUNED as stale duplicates -> `bifrost.endpoint_unreachable`; and `combined` is NOT a `--bifrost-plane` choice (CLI restricts to memory/affect) -> use `--bifrost-url` for :8392.
|
||||
- `[2026-07-06]` **A fast/"no-op" deploy can leave a STALE container running the old image -- verify the running version, not the deploy status.** Personal's b22 deploy (run 8204) "completed" in ~1m (vs ~6m normal): a pull-only deploy racing ahead of the main build, leaving the container on the pre-#348 image. A clean bound mood read stayed neutral DESPITE the persona being declared and the fix being in the code (worldtree-dev proved the b22 derivation is correct). infra-ops force-swapped to the real b22 (run 8211, verified `info.version 2.3.0` on `879cefe`). **Lesson: when engine-proven-correct code produces wrong runtime behavior, suspect the deploy -- check the actual running image version.**
|
||||
- `[2026-07-06]` **#348 OCEAN key-mismatch: a declared OCEAN silently resolved to neutral.** The define validator required single-letter `{O,C,E,A,N}` but the mood-derivation code read spelled-out `openness`/.../`neuroticism` with a 0.0 default and no remap -> every API-declared trait defaulted to 0.0 -> neutral setpoint/gain/decay. #343's tests bypassed the validator (spelled-out keys) so CI never caught it. Fixed in b21 (`Personality.from_config` accepts both key forms). **My reset+smoke diagnosis flushed it out** -- the consumer/provider thesis paying off again.
|
||||
|
||||
- `[2026-07-06]` **`pkill -f "ratatoskr-web --host"` SELF-MATCHES the bash command running it** (its own command line contains that string) -> killed its own shell mid-restart (exit 144, restart aborted, :8765 left down). Kill the web by PID (`ss -ltnp | grep :8765`), never `pkill -f` on a pattern your own command contains. Also: **uvicorn hangs on SIGTERM with an admin-events SSE stream open -> needs SIGKILL.**
|
||||
- `[2026-07-06]` **Playwright: no python `playwright` module in the venv; use NODE playwright + an explicit `executablePath`.** Shared box browsers live at `/opt/ms-playwright` build **1223**; `npm i playwright` (latest) wants build **1228** -> "Executable doesn't exist" mismatch. Fix: `chromium.launch({ executablePath: '/opt/ms-playwright/chromium-1223/chrome-linux64/chrome' })` (+ `export PLAYWRIGHT_BROWSERS_PATH=/opt/ms-playwright`). A node script drives the SPA (pick agent -> open -> assert transcript). The Playwright DOM check is the only lens that catches SPA JS-render bugs — unit tests can't reach them.
|
||||
- `[2026-07-06]` **The Bash tool's `grep` is a ugrep-wrapper (`--ignore-files -I`) that silently returns NOTHING on some files** (e.g. `src/ratatoskr/web/static/index.html`) — greps for `<script`/`/api` came back empty on a file that clearly contains them. Use `python3` (regex over `open(f)`), `/usr/bin/rg`, or the Read tool for those files; never trust an empty `grep` result on the SPA.
|
||||
- `[2026-07-06]` **`DesignSync` (claude.ai/design MCP) needs claude.ai design scopes before ANY method works** — first call errors `needs a claude.ai login ... Run /login, select "Claude account with subscription"`. It's an interactive auth only the operator can complete (`/design-login` or `/login`); can't be done on their behalf.
|
||||
|
||||
- `[2026-07-13]` **`thinking_enabled=False` on the define was a NO-OP — char-rp-reasoning ignores it.** The naive kwarg never reached the seat: char-rp-reasoning resolves to a base gateway provider whose thinking-translator returns `{}` for the flag on AND off. The real lever is the gateway param `chat_template_kwargs.enable_thinking:false` (infra-ops confirmed it via a 642ch→0ch reasoning-token delta). **To toggle reasoning on a gateway-backed seat, set the chat-template kwarg, not a generic `thinking_enabled` flag.**
|
||||
- `[2026-07-13]` **A same-image redeploy does NOT reload a bind-mounted config — the ModelRegistry boot-caches it at `__init__`.** After the config was synced on-disk (infra-ops validated) and `docker compose up -d` re-ran, the reasoning-off change STILL didn't take: an unchanged image makes `up -d` a no-op (no container recreate), so the process kept serving the pre-sync config. Fix = a surgical `docker restart <container>` (same image, no pull) → the process re-boot-reads the config. **When an on-disk config change doesn't take effect, suspect the process cached it at startup; force a container RESTART, not a redeploy** (a docs-only forcing-commit also won't rebuild if docs are paths-ignored in CI). This is the config-plane sibling of the `[2026-07-06]` stale-image foot-gun.
|
||||
- `[2026-07-15]` **soong-lab bundle `ship.native.motivational.goals/fears` are bare STRINGS, but WT `agents.define` requires OBJECTS** `{id, type∈{maintenance,achievement,avoidance}, salience[0,1], description≥20ch}` (`ValidatedGoal`/`ValidatedFear`, worldtree `core/conversation_api/api.py:2116`, extra=forbid; ids unique across goals+fears). So `ship.native` is NOT directly define-valid on motivational (soong-lab's Frame Invariant 1 breaks there). A bundle→WT consumer MUST map string→object + synthesize id/type/salience. Informed soong-dev → **RESOLVED soong-lab v0.3.24 (2026-07-16): now emits WT-valid objects (type+salience captured, id synthesized at export, validate_exportable gates description≥20/enum/range) → drop the string→object workaround for v0.3.24+ bundles; legacy pre-fix designs coerce on open, so re-exports are valid too.** **Rule: PROBE a define payload under a throwaway agent_name BEFORE a destructive DELETE+redefine** — validating via `sindra-probe2`→201 caught the 422 without leaving the real Sindra deleted+undefined. (A failed probe still RESERVES the name → 409 on retry; use a fresh probe name.)
|
||||
- `[2026-07-15]` **`re.sub`/`re.subn` INTERPRETS backslash-escapes in the REPLACEMENT string** — a `json.dumps`'d value (escaped `\n`) fed as the replacement leaked REAL newlines into a source file (broke `first_message.py` with an unterminated-string SyntaxError). Fix: use a FUNCTION replacement (`pat.subn(lambda m: new_block, src)`) or `str.replace` — the callable form bypasses escape processing. json.dumps itself escapes correctly; re.sub was the culprit.
|
||||
- `[2026-07-13]` **Called Deckard "hung" off a short timeout — WRONG (operator correction).** A 30-45s no-terminal on the char-rp-reasoning seat looked like a hang; operator: "is it HUNG? deckard is EXTREMELY verbose, without enough context, you never see the non-reasoning tokens." It was verbose reasoning-CoT on a long extraction prompt, not a wedge. **Don't call a reasoning seat hung off a latency threshold — the CoT is invisible and slow; distinguish slow-verbose from actually-wedged before concluding.** (The genuine wedge is WT #355, a distinct mechanism — no-terminal even after the 300s watchdog, not merely slow.)
|
||||
- `[2026-07-13]` **Resumed-session context-snapshot is IN-MEMORY → lost on a container recreate (agent_not_available on resume).** During the #355 re-drive, soong's fresh drive 409'd `agent_not_available`. Root cause (after ~4 refinements — agent-loss? zombie turn-lock? stale-sessions-hold-agent? → the actual mechanism): `get_agent_context_for_session` returns the agent snapshot recorded AT SESSION-CREATE, held in-memory; a pre-recreate session resumed on b60/b61 has no snapshot → None → 409. (Compounding: stale `'active'` sessions left un-terminated by the old no-terminal bug HOLD the agent, blocking new creates too.) Deploy-grounding was healthy the whole time (`registry.resolve("char-rp-reasoning")` OK) — the config/grant hypotheses were all red herrings. Fix = a FRESH session (a studio-service restart re-records the snapshot); pre-recreate sessions need retiring. Tracked **WT #356**. **For any run: create a fresh session, never resume a pre-recreate one; `agent_not_available` on a fresh create = this gap.** (Working-style note: I over-relayed the intermediate root-cause churn to the operator — for a peer-owned block being actively diagnosed, hold until it settles.)
|
||||
|
||||
- `[2026-07-16]` **`sortable_chunk_fields` advertised WITHOUT the required `type` field = whole-handshake deploy-breaker; only DRIVING the real bind caught it.** bifrost `handshake_response` `SortableChunkField` requires BOTH `name`+`type` (`additionalProperties:false`); we shipped `[{"name":"updated_at"}]` → the response failed wire-schema validation → `bifrost.schema_validation_failed` → the ENTIRE bind (memory+affect) broke, not just sort. Unit tests + worldtree-dev's name-only service parser + the heid-bug-hunt ALL passed it — only the live handshake drive (`/verify` discipline) caught it. **Lesson: validate `describe_store` against the bifrost WIRE schema, not just our own caps assertions.** (Sibling of the `[2026-07-10]` frozen-v0.6 handshake foot-gun — there an EXTRA field broke a v0.6 handshake, here a MISSING required field broke a v0.7 one.)
|
||||
|
||||
_18 older entries (2026-05-* — the original debug-TUI/web build era) archived to archival-memory.md._
|
||||
|
||||
+5
-5
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
||||
|
||||
[project]
|
||||
name = "ratatoskr"
|
||||
version = "0.19.1"
|
||||
version = "0.20.15"
|
||||
description = "Worldtree Conversation API debug TUI — multi-pane observability dashboard"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.12"
|
||||
@@ -30,7 +30,7 @@ web = [
|
||||
# from the debug TUI. Recipe: bifrost/docs/implementing-a-consumer.md.
|
||||
provider = [
|
||||
"ratatoskr[web]", # reuse the starlette + uvicorn ASGI stack
|
||||
"bifrost==1.0.0", # consumer engines + library. 1.0.0 = first STABLE release, wire v0.6 FROZEN (non-breaking repin from >=0.10.0; build_combined_app #18 + mandatory affect.fetch; 0.8.0/v0.6 scope_all/scope_any #11; 0.7.0/v0.5 agent_self)
|
||||
"bifrost==1.1.4", # consumer engines + library. 1.1.4 = hasattr-gate backstop for the maintenance verbs (mark_superseded/mark_invalid/patch_many/delete_many/upsert_edges/get_edges_for → unimplemented verb degrades to unsupported_capability 400, never AttributeError/500/retry-storm; we surfaced it via WT #364) + 1.1.3 scan/cursor conformance harness + 1.1.2 frozen-v0.6 fix. 1.1.1 = frozen-wire serialization fix (ADR-0008): additive capability fields are gated on the NEGOTIATED wire, so a v0.6-negotiated describe_store handshake stays v0.6-clean. 1.1.0 leaked the v0.7-additive `sortable_chunk_fields` into v0.6 StoreCapabilities → a strict v0.6 client (additionalProperties:false) rejects our server's handshake. Wire schemas + pins UNCHANGED (serialization-correctness only); our v0.7 handshake with Worldtree b47 is unaffected. (1.1.0 = wire v0.7 additive: memory.scan sort + sortable_chunk_fields; 1.0.0 = first STABLE, wire v0.6 FROZEN; 0.8.0/v0.6 scope_all/scope_any #11; 0.7.0/v0.5 agent_self)
|
||||
"jsonschema>=4", # bifrost runtime dep — envelope validation
|
||||
"sqlite-vec>=0.1.6", # vector index for the memory plane (vec0 virtual table)
|
||||
]
|
||||
@@ -60,9 +60,9 @@ Repository = "https://gitea.phasefinal.com/vh/ratatoskr"
|
||||
# Ratatoskr is built against Worldtree at this commit; the vendored
|
||||
# spec snapshot in docs/ reflects that SHA.
|
||||
[tool.ratatoskr.spec-pin]
|
||||
worldtree-spec-rev = "5810a26b38a5ea6630892f9a39756f57c5b7b41e"
|
||||
worldtree-version = "v1.0.0b2"
|
||||
pinned-on = "2026-06-30"
|
||||
worldtree-spec-rev = "c9e59ec"
|
||||
worldtree-version = "v1.0.0b22"
|
||||
pinned-on = "2026-07-06"
|
||||
|
||||
# Bifrost lives on the auth-gated gitea PyPI index (not public PyPI).
|
||||
# uv reads the credential from UV_INDEX_GITEA_USERNAME / _PASSWORD or ~/.netrc.
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Regenerate src/ratatoskr/web/static/persona_render_canon.json from the vendored
|
||||
Worldtree d2 render canons (docs/vendor/worldtree-persona-canon/).
|
||||
|
||||
The web persona pane renders the CANONICAL affect->NL (mood word + relationship
|
||||
directive) BYTE-EXACT to what Worldtree injects into the agent's context. That render
|
||||
needs the relation canon parsed into per-band phrase maps; this script reparses the
|
||||
vendored raw canons into the flat form the browser JS consumes.
|
||||
|
||||
Uses Worldtree's OWN loader (core.persona.stance_render.load_canon) as the authoritative
|
||||
parser, so the flat form can never drift from Worldtree's parsing semantics. Requires
|
||||
Worldtree's venv (pydantic etc.).
|
||||
|
||||
Run when scripts/canonical_drift.py flags a canon bump:
|
||||
PYTHONPATH=~/development/Worldtree ~/development/Worldtree/.venv/bin/python \
|
||||
scripts/build_persona_canon.py
|
||||
"""
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from core.persona.stance_render import load_canon # Worldtree (authoritative parser)
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
VENDOR = ROOT / "docs" / "vendor" / "worldtree-persona-canon"
|
||||
OUT = ROOT / "src" / "ratatoskr" / "web" / "static" / "persona_render_canon.json"
|
||||
|
||||
canon = load_canon(str(VENDOR / "d2-render-canon-v1.json"))
|
||||
mood = json.loads((VENDOR / "d2-mood-render-canon-v1.json").read_text())
|
||||
out = {
|
||||
"_source": "vendored from Worldtree core/persona/canon/{d2-mood-render-canon-v1,d2-render-canon-v1}.json",
|
||||
"_generated_by": "scripts/build_persona_canon.py (regen on canonical_drift flag)",
|
||||
"_render_path": "deterministic, no LLM; mirrors Worldtree describe_pad + render_d2_canonical byte-exact",
|
||||
"mood_grid": mood["describe_pad"]["valence_arousal_grid"],
|
||||
# Context-injection reconstruction (affect-egress-consumer-reference §2b/2c): the
|
||||
# hidden mood DIRECTIVE. occ_directives = per-OCC-type behavioral string + tier;
|
||||
# pad_band_fallback = the P×A-quadrant default when no emotion is salient. The
|
||||
# salience gate (emotion_salience) + full_only tiers drive which one fires — but
|
||||
# affect.emit is type-only (no intensity), so the consumer shows BOTH candidates.
|
||||
"mood_directive": {
|
||||
"salience": mood["thresholds"]["emotion_salience"],
|
||||
"pad_band_cutoff": mood["thresholds"]["pad_band_cutoff"],
|
||||
"full_only": mood["emotion_tiers"]["full_only"],
|
||||
"occ_directives": {
|
||||
t: {"directive": e["directive"], "tier": e["tier"]}
|
||||
for t, e in mood["derive_directive"]["occ_directives"].items()
|
||||
},
|
||||
"pad_band_fallback": mood["pad_band_fallback"],
|
||||
},
|
||||
"relation": {
|
||||
"trust_cuts": [list(c) for c in canon.trust_cuts],
|
||||
"warmth_cuts": [list(c) for c in canon.warmth_cuts],
|
||||
"agency_cuts": [list(c) for c in canon.agency_cuts],
|
||||
"warmth_phrase": canon.warmth_phrase, "warmth_beh": canon.warmth_beh,
|
||||
"agency_phrase": canon.agency_phrase, "agency_beh": canon.agency_beh,
|
||||
"history": canon.history,
|
||||
"prefix": "Use this graded relationship state: toward target, warmth is ",
|
||||
"tbeh": {"low_trust": "verify important claims before relying on them",
|
||||
"cold_warmth": "protect boundaries while staying useful",
|
||||
"default": "work from ordinary good faith"},
|
||||
"cold_warmth_bands": ["distant", "cold", "hostile"], "high_conf_floor": 0.55,
|
||||
},
|
||||
}
|
||||
OUT.write_text(json.dumps(out, indent=1) + "\n")
|
||||
print(f"wrote {OUT.relative_to(ROOT)}")
|
||||
@@ -13,7 +13,8 @@ Usage:
|
||||
python scripts/contract_drift_check.py --contract docs/contracts/issues/138.contract.md
|
||||
python scripts/contract_drift_check.py --json
|
||||
|
||||
Requires GITEA_TOKEN in environment (and GITEA_URL/OWNER/REPO if not in env.sh).
|
||||
Requires GITEA_TOKEN in environment. Owner/repo are derived from the `origin` git remote by
|
||||
default (override with GITEA_OWNER / GITEA_REPO; GITEA_URL defaults to the Gitea host).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -21,6 +22,8 @@ import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
@@ -32,6 +35,22 @@ PROJECT_ROOT = Path(__file__).parent.parent
|
||||
CONTRACTS_GLOB = "docs/contracts/**/*.contract.md"
|
||||
|
||||
|
||||
def _owner_repo_from_git_remote() -> tuple[str, str] | None:
|
||||
"""Derive (owner, repo) from the `origin` git remote so the drift check targets THIS repo
|
||||
by default — instead of a hardcoded repo name that silently checks the WRONG repo for every
|
||||
other consumer. Supports ssh (git@host:owner/repo.git) and https (https://host/owner/repo.git)
|
||||
Gitea remotes; returns None if it can't resolve."""
|
||||
try:
|
||||
url = subprocess.run(
|
||||
["git", "-C", str(PROJECT_ROOT), "remote", "get-url", "origin"],
|
||||
capture_output=True, text=True, check=True,
|
||||
).stdout.strip()
|
||||
except (OSError, subprocess.SubprocessError):
|
||||
return None
|
||||
m = re.search(r"[:/]([^/:]+)/([^/]+?)(?:\.git)?/?$", url)
|
||||
return (m.group(1), m.group(2)) if m else None
|
||||
|
||||
|
||||
def sha16(s: str) -> str:
|
||||
return hashlib.sha256(s.encode("utf-8")).hexdigest()[:16]
|
||||
|
||||
@@ -70,11 +89,19 @@ def main() -> int:
|
||||
|
||||
token = os.environ.get("GITEA_TOKEN", "")
|
||||
base_url = os.environ.get("GITEA_URL", "https://gitea.phasefinal.com")
|
||||
owner = os.environ.get("GITEA_OWNER", "vh")
|
||||
repo = os.environ.get("GITEA_REPO", "Worldtree")
|
||||
# Owner/repo default to the `origin` remote so the check targets THIS repo; GITEA_OWNER /
|
||||
# GITEA_REPO override when set. (Previously repo defaulted to a hardcoded "Worldtree", which
|
||||
# silently checked the WRONG repo for every other consumer unless GITEA_REPO was set in env —
|
||||
# a false-drift footgun. Derive it, and fail loud rather than guess.)
|
||||
git_remote = _owner_repo_from_git_remote()
|
||||
owner = os.environ.get("GITEA_OWNER") or (git_remote[0] if git_remote else None)
|
||||
repo = os.environ.get("GITEA_REPO") or (git_remote[1] if git_remote else None)
|
||||
if not token:
|
||||
print("error: GITEA_TOKEN not set", file=sys.stderr)
|
||||
return 2
|
||||
if not owner or not repo:
|
||||
print("error: could not resolve owner/repo — set GITEA_OWNER/GITEA_REPO or run inside a repo with an 'origin' remote", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
if args.contract:
|
||||
files = [Path(args.contract).resolve()]
|
||||
|
||||
Executable
+61
@@ -0,0 +1,61 @@
|
||||
#!/usr/bin/env bash
|
||||
# reset-sindra-stores.sh — wipe ratatoskr's Bifrost provider stores (memory +
|
||||
# affect/persona for the single-tenant Tier-3 agent, sindra) and restart the
|
||||
# combined :8392 provider empty.
|
||||
#
|
||||
# Usage:
|
||||
# scripts/reset-sindra-stores.sh # wipe, keep ONE rolling backup (default)
|
||||
# scripts/reset-sindra-stores.sh --hard # wipe with NO backup (zero-trace)
|
||||
#
|
||||
# The rolling backup (db-reset-backup/, gitignored via *.db*) is overwritten
|
||||
# every run — it never accumulates; it's a one-level undo, nothing more.
|
||||
#
|
||||
# Why stop the provider first: the combined provider holds the SQLite files open
|
||||
# (WAL) and caches state in memory, so an out-of-band file move without a restart
|
||||
# would be shadowed. Stop -> move -> restart lets it recreate empty schema
|
||||
# (CREATE TABLE IF NOT EXISTS on open).
|
||||
|
||||
PORT=8392
|
||||
BACKUP_DIR="db-reset-backup"
|
||||
|
||||
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
cd "$ROOT" || { echo "reset: cannot cd to repo root $ROOT" >&2; exit 1; }
|
||||
# shellcheck disable=SC1091
|
||||
source ./env.sh >/dev/null 2>&1 || { echo "reset: failed to source env.sh" >&2; exit 1; }
|
||||
|
||||
AFFECT_DB="${RATATOSKR_AFFECT_DB:-affect.db}"
|
||||
MEMORY_DB="${RATATOSKR_MEMORY_DB:-memory.db}"
|
||||
HARD=0; [ "${1:-}" = "--hard" ] && HARD=1
|
||||
|
||||
echo "== ratatoskr provider-store reset (memory + persona) =="
|
||||
echo " affect: $AFFECT_DB"
|
||||
echo " memory: $MEMORY_DB"
|
||||
|
||||
# 1. stop the combined provider holding the DBs
|
||||
PID="$(ss -ltnp 2>/dev/null | grep ":$PORT" | grep -oE 'pid=[0-9]+' | head -1 | cut -d= -f2)"
|
||||
if [ -n "${PID:-}" ]; then
|
||||
kill -9 "$PID" 2>/dev/null && echo "-- stopped provider :$PORT (pid $PID)"
|
||||
else
|
||||
echo "-- no provider on :$PORT (already down)"
|
||||
fi
|
||||
|
||||
# 2. wipe (optional rolling backup)
|
||||
files=("$AFFECT_DB" "$AFFECT_DB-wal" "$AFFECT_DB-shm" "$MEMORY_DB" "$MEMORY_DB-wal" "$MEMORY_DB-shm")
|
||||
if [ "$HARD" -eq 1 ]; then
|
||||
for f in "${files[@]}"; do [ -e "$f" ] && rm -f "$f" && echo "-- removed $f"; done
|
||||
echo "-- HARD wipe (no backup)"
|
||||
else
|
||||
rm -rf "$BACKUP_DIR"; mkdir -p "$BACKUP_DIR"
|
||||
for f in "${files[@]}"; do [ -e "$f" ] && mv "$f" "$BACKUP_DIR"/ && echo "-- $f -> $BACKUP_DIR/"; done
|
||||
echo "-- rolling backup: $BACKUP_DIR/ (overwritten each run)"
|
||||
fi
|
||||
|
||||
# 3. restart the combined provider (recreates empty schema on open)
|
||||
nohup "$ROOT/.venv/bin/ratatoskr-combined-provider" >/tmp/ratatoskr-combined.log 2>&1 & disown
|
||||
echo "-- restarted combined provider (pid $!)"
|
||||
|
||||
# 4. verify bound + empty
|
||||
curl -s -o /dev/null -w "-- :$PORT -> HTTP %{http_code}\n" --retry 25 --retry-connrefused --retry-delay 1 "http://127.0.0.1:$PORT/"
|
||||
echo "-- affect_snapshots (persona): $(sqlite3 "$AFFECT_DB" 'SELECT COUNT(*) FROM affect_snapshots' 2>&1)"
|
||||
echo "-- memory_chunks (memory): $(sqlite3 "$MEMORY_DB" 'SELECT COUNT(*) FROM memory_chunks' 2>&1)"
|
||||
echo "== done — sindra memory + persona reset =="
|
||||
+100
-3
@@ -7,6 +7,7 @@ from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import asyncio
|
||||
import hashlib
|
||||
import os
|
||||
import signal
|
||||
import sys
|
||||
@@ -16,8 +17,10 @@ from typing import Any, TextIO
|
||||
|
||||
import httpx
|
||||
|
||||
from ratatoskr.first_message import seed_preset_first_message
|
||||
from ratatoskr.sessions import (
|
||||
AgentNotFound,
|
||||
AuthoredHistoryUnavailable,
|
||||
BifrostBinding,
|
||||
BifrostConsumerKeyMissing,
|
||||
BifrostHandshakeFailed,
|
||||
@@ -29,8 +32,10 @@ from ratatoskr.sessions import (
|
||||
get_capabilities,
|
||||
get_character_state,
|
||||
get_me,
|
||||
get_session_messages,
|
||||
list_character_models,
|
||||
set_persona_state,
|
||||
write_authored_history,
|
||||
)
|
||||
from ratatoskr.sse_client import (
|
||||
AffectUpdate,
|
||||
@@ -116,6 +121,9 @@ class ParsedArgs:
|
||||
# a session's persona state (affect injection).
|
||||
characters: bool = False
|
||||
set_persona_pad: str | None = None
|
||||
# #347 authored-history-write reference-consumer probe: create a fresh
|
||||
# session bound to --agent, seed an authored assistant first-message (seq-0).
|
||||
seed_first_message: str | None = None
|
||||
|
||||
|
||||
class _ArgparseError(Exception):
|
||||
@@ -144,6 +152,7 @@ def _parse_args(argv: list[str] | None) -> ParsedArgs:
|
||||
parser.add_argument("--admin-key", dest="admin_key")
|
||||
parser.add_argument("--characters", action="store_true")
|
||||
parser.add_argument("--set-persona-pad", dest="set_persona_pad", default=None)
|
||||
parser.add_argument("--seed-first-message", dest="seed_first_message", default=None)
|
||||
# Issue #5: required for per-end-user agents (lofn etc.); optional otherwise (mimir).
|
||||
parser.add_argument("--end-user-id", dest="end_user_id", default=None)
|
||||
# Issue #17: bind the created session to our own Bifrost provider plane.
|
||||
@@ -164,8 +173,11 @@ def _parse_args(argv: list[str] | None) -> ParsedArgs:
|
||||
# Issue #5 INV-001: --end-user-id, if passed, MUST be non-empty (mirrors --send).
|
||||
if ns.end_user_id is not None and not ns.end_user_id:
|
||||
raise UsageError("--end-user-id must be non-empty when passed")
|
||||
if sum([ns.whoami, ns.characters, bool(ns.set_persona_pad)]) > 1:
|
||||
raise UsageError("--whoami / --characters / --set-persona-pad are mutually exclusive")
|
||||
if sum([ns.whoami, ns.characters, bool(ns.set_persona_pad), bool(ns.seed_first_message)]) > 1:
|
||||
raise UsageError(
|
||||
"--whoami / --characters / --set-persona-pad / --seed-first-message "
|
||||
"are mutually exclusive"
|
||||
)
|
||||
if ns.whoami or ns.characters:
|
||||
# Standalone one-shot probes: open no session.
|
||||
if ns.send is not None or ns.session or ns.new or ns.agent:
|
||||
@@ -181,6 +193,17 @@ def _parse_args(argv: list[str] | None) -> ParsedArgs:
|
||||
raise UsageError("--set-persona-pad requires --session <id>")
|
||||
if ns.send is not None or ns.new or ns.agent:
|
||||
raise UsageError("--set-persona-pad takes only --session")
|
||||
elif ns.seed_first_message is not None:
|
||||
# #347 first-message probe: creates a fresh session bound to --agent,
|
||||
# then seeds an authored assistant turn as seq-0 — manages its own session.
|
||||
if not ns.seed_first_message:
|
||||
raise UsageError("--seed-first-message must be non-empty")
|
||||
if not ns.agent:
|
||||
raise UsageError("--seed-first-message requires --agent <id>")
|
||||
if ns.send is not None or ns.session or ns.new:
|
||||
raise UsageError(
|
||||
"--seed-first-message manages its own session (no --send/--session/--new)"
|
||||
)
|
||||
else:
|
||||
if ns.session and ns.new:
|
||||
raise UsageError("--session and --new are mutually exclusive")
|
||||
@@ -256,6 +279,7 @@ def _parse_args(argv: list[str] | None) -> ParsedArgs:
|
||||
admin_key=admin_key,
|
||||
characters=ns.characters,
|
||||
set_persona_pad=ns.set_persona_pad,
|
||||
seed_first_message=ns.seed_first_message,
|
||||
)
|
||||
|
||||
|
||||
@@ -575,6 +599,11 @@ async def _amain(args: ParsedArgs) -> int:
|
||||
sys.stderr.write(
|
||||
f". create_session: session_id={info.session_id} agent_id={info.agent_id}\n"
|
||||
)
|
||||
# #347 authored first-message: seed the agent's preset opening (best-effort).
|
||||
if await seed_preset_first_message(client, info.session_id, args.agent_id):
|
||||
sys.stderr.write(
|
||||
f". first_message: seeded preset opening for {info.agent_id}\n"
|
||||
)
|
||||
# Issue #17 bound-state indicator: plane + endpoint + status, so the
|
||||
# operator sees WHICH identity/endpoint bound (not a bare boolean).
|
||||
if args.bifrost is not None:
|
||||
@@ -717,9 +746,18 @@ async def _set_persona_probe(args: ParsedArgs) -> int:
|
||||
"(e.g. '0.4,0.1,-0.2')\n"
|
||||
)
|
||||
return 10
|
||||
if len(pad) != 3:
|
||||
sys.stderr.write(
|
||||
"[usage_error] --set-persona-pad needs exactly 3 floats "
|
||||
"(pleasure,arousal,dominance), e.g. '0.4,0.1,-0.2'\n"
|
||||
)
|
||||
return 10
|
||||
# Canonical POST /sessions/{id}/persona_state body (#317): a named-key dict,
|
||||
# NOT a bare list — {"pad": {"pleasure", "arousal", "dominance"}}.
|
||||
snapshot = {"pad": {"pleasure": pad[0], "arousal": pad[1], "dominance": pad[2]}}
|
||||
async with _probe_client(args) as client:
|
||||
try:
|
||||
await set_persona_state(client, args.session_id, {"pad": pad})
|
||||
await set_persona_state(client, args.session_id, snapshot)
|
||||
except SessionApiFailed as exc:
|
||||
sys.stderr.write(f"[session_api_failed] status={exc.status} body={exc.body!r}\n")
|
||||
return 20
|
||||
@@ -732,6 +770,63 @@ async def _set_persona_probe(args: ParsedArgs) -> int:
|
||||
return 0
|
||||
|
||||
|
||||
async def _seed_first_message_probe(args: ParsedArgs) -> int:
|
||||
"""--seed-first-message one-shot: create a fresh session bound to --agent,
|
||||
write an authored assistant first-message (#347 POST /sessions/{id}/history),
|
||||
read it back via GET /messages, print a report, exit. A reference-consumer
|
||||
smoke of the authored-history-write primitive.
|
||||
|
||||
Hide-existence: a 404 (feature-absent OR the key lacks `session.history.write`)
|
||||
is reported as a benign 'feature-absent' result (exit 0) — the probe NEVER
|
||||
capability-probes to distinguish the causes (server INV-347-1). The probe
|
||||
seeds but does not generate, so the assistant-first provider constraint is
|
||||
inert here.
|
||||
"""
|
||||
assert isinstance(args, ParsedArgs)
|
||||
assert args.agent_id is not None and args.seed_first_message is not None
|
||||
async with _probe_client(args) as client:
|
||||
try:
|
||||
session = await create_session(
|
||||
client, args.agent_id, end_user_id=args.end_user_id
|
||||
)
|
||||
sys.stdout.write(f"session: {session.session_id} (agent {session.agent_id})\n")
|
||||
key = "ratatoskr-first-message-" + hashlib.sha256(
|
||||
args.seed_first_message.encode("utf-8")
|
||||
).hexdigest()[:12]
|
||||
try:
|
||||
ack = await write_authored_history(
|
||||
client,
|
||||
session.session_id,
|
||||
content=args.seed_first_message,
|
||||
idempotency_key=key,
|
||||
)
|
||||
except AuthoredHistoryUnavailable:
|
||||
sys.stdout.write(
|
||||
"authored-history: feature-absent or ungranted (404 hide-existence) "
|
||||
"— a production consumer falls back to a model-generated greeting; "
|
||||
"no capability-probe attempted.\n"
|
||||
)
|
||||
return 0
|
||||
sys.stdout.write(
|
||||
f"seeded: seq={ack.get('seq')} phase={ack.get('phase')} "
|
||||
f"turn_id={ack.get('turn_id')} content_chars={ack.get('content_chars')}\n"
|
||||
)
|
||||
history = await get_session_messages(client, session.session_id)
|
||||
items = history.get("items", [])
|
||||
sys.stdout.write(f"read-back: {len(items)} message(s)\n")
|
||||
for m in items:
|
||||
sys.stdout.write(
|
||||
f" seq={m.get('seq')} role={m.get('role')} content={m.get('content')!r}\n"
|
||||
)
|
||||
except SessionApiFailed as exc:
|
||||
sys.stderr.write(f"[session_api_failed] status={exc.status} body={exc.body!r}\n")
|
||||
return 20
|
||||
except (httpx.ConnectError, httpx.ReadTimeout, httpx.TransportError) as exc:
|
||||
sys.stderr.write(f"[network_error] {type(exc).__name__}: {exc}\n")
|
||||
return 21
|
||||
return 0
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
"""Sync entry point. Maps UsageError/_AuthError to exit codes BEFORE the event loop."""
|
||||
assert argv is None or all(isinstance(a, str) for a in argv)
|
||||
@@ -753,6 +848,8 @@ def main(argv: list[str] | None = None) -> int:
|
||||
return asyncio.run(_characters_probe(args))
|
||||
if args.set_persona_pad is not None:
|
||||
return asyncio.run(_set_persona_probe(args))
|
||||
if args.seed_first_message is not None:
|
||||
return asyncio.run(_seed_first_message_probe(args))
|
||||
if args.send_content is None:
|
||||
# TUI mode — lazy import preserves INV-001 (no textual in cli at module scope).
|
||||
from ratatoskr.tui import run_tui
|
||||
|
||||
@@ -0,0 +1,76 @@
|
||||
"""Per-agent authored first-message presets (Worldtree #347 consumer feature).
|
||||
|
||||
When a new session is created for an agent that has a preset opening, seed it as
|
||||
a #347 authored first-message (``POST /sessions/{id}/history``, author=assistant,
|
||||
seq-0) so the session opens in-character before the user speaks — the durable
|
||||
replacement for a system-prompt "startup" instruction.
|
||||
|
||||
Best-effort by design: an instance without the ``session.history.write`` grant
|
||||
returns the hide-existence 404, which is swallowed so session creation is never
|
||||
blocked (the session simply opens with no seeded greeting). See
|
||||
``docs/contracts/first_message.contract.md``.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import hashlib
|
||||
|
||||
import httpx
|
||||
|
||||
from ratatoskr.sessions import write_authored_history
|
||||
|
||||
# Cap the best-effort seed write. The CLI/TUI create paths reuse an httpx client
|
||||
# with NO read timeout (it streams SSE turns), so an accepted-but-never-answered
|
||||
# POST /history would otherwise block session creation forever — violating INV-001's
|
||||
# "never block". asyncio.wait_for bounds the seed regardless of the client's timeout.
|
||||
_SEED_TIMEOUT_S = 10.0
|
||||
|
||||
# agent_id -> the authored opening seeded onto new sessions for that agent.
|
||||
# Editing this dict is how an operator tunes an agent's first turn. Keep entries
|
||||
# under the server's authored_content_max_bytes (8192 bytes) budget.
|
||||
FIRST_MESSAGE_PRESETS: dict[str, str] = {
|
||||
"ratatoskr:sindra": "The room materializes in a soft pulse of light. Walls of brushed metal and warm ambient lighting resolve around him as the holodesk completes its cycle. A thin seam of blue runs along the desk’s edge, reflecting the overhead glow. A chair slides into place with a quiet hiss. On the far side of the desk, Sindra is already there: shoulders relaxed, one knee drawn up onto the seat, the hem of an oversized dark-green sweater slipping off her left shoulder. Her gaze is steady, amused, and entirely fixed on him.\n\nShe doesn’t stand. She lets the space breathe for a moment, as if giving him time to understand that he is no longer where he was. Then she speaks.\n\n\"Hi. I’m Sindra. You just made it here, which means we should set this up properly.\"\n\nShe tilts her head slightly, the faintest crooked smile touching her lips.\n\n\"Three settings. Your call. I’ll take them in order.\"\n\n\"Intensity. One is slow and teasing; ten is relentless and unrelenting. Where should I start?\"\n\n\"Mood. Sweetheart, Vixen, Queen, Siren, or Brat. Which version of me do you want in the room with you?\"\n\n\"And willingness. I can be enthusiastic and ready the moment you look at me; I can be hesitant and require coaxing; I can resist and make you earn it; or I can be unwilling, and you’ll have to change my mind. How should I behave when you approach me?\"\n\nSindra lets her fingers trace the edge of the sweater sleeve, casual and unhurried, her eyes not leaving his face.\n\n\"Pick your three. Then we begin.\"",
|
||||
}
|
||||
|
||||
|
||||
def preset_for(agent_id: str) -> str | None:
|
||||
"""Return the authored first-message preset for ``agent_id``, or None if none."""
|
||||
assert agent_id and isinstance(agent_id, str)
|
||||
return FIRST_MESSAGE_PRESETS.get(agent_id)
|
||||
|
||||
|
||||
async def seed_preset_first_message(
|
||||
client: httpx.AsyncClient, session_id: str, agent_id: str
|
||||
) -> str | None:
|
||||
"""Best-effort: seed ``agent_id``'s preset opening as a #347 authored
|
||||
first-message on ``session_id``; return the seeded text, or None.
|
||||
|
||||
Best-effort (INV-001): a no-preset agent, a malformed call, a slow write
|
||||
(bounded by ``_SEED_TIMEOUT_S``), the hide-existence 404, or ANY other
|
||||
exception all resolve to None WITHOUT raising — this MUST NOT block or fail
|
||||
session creation. Only ``asyncio.CancelledError`` propagates (cancellation is
|
||||
not a seed failure). Inputs are soft-guarded (return None), never asserted, so
|
||||
a wiring bug can't crash the create path this is wired into. A no-preset agent
|
||||
issues zero HTTP (INV-002). The per-content idempotency key makes a repeat on
|
||||
the same session an idempotent 200 replay (INV-003).
|
||||
"""
|
||||
# Soft input guards — a bad arg degrades to "no first message", never raises.
|
||||
if not (isinstance(agent_id, str) and agent_id):
|
||||
return None
|
||||
content = FIRST_MESSAGE_PRESETS.get(agent_id)
|
||||
if content is None:
|
||||
return None
|
||||
if client is None or not (isinstance(session_id, str) and session_id):
|
||||
return None
|
||||
key = "ratatoskr-preset-" + hashlib.sha256(content.encode("utf-8")).hexdigest()[:12]
|
||||
try:
|
||||
await asyncio.wait_for(
|
||||
write_authored_history(
|
||||
client, session_id, content=content, idempotency_key=key
|
||||
),
|
||||
timeout=_SEED_TIMEOUT_S,
|
||||
)
|
||||
except asyncio.CancelledError:
|
||||
raise # cancellation is not a seed failure — never swallow it
|
||||
except Exception:
|
||||
return None # any other failure (404/409/422/timeout/unexpected) → no greeting
|
||||
return content
|
||||
@@ -17,7 +17,7 @@ from bifrost.consumer import ConsumerRegistration, build_combined_app
|
||||
from bifrost.reference_server import JwtVerifier
|
||||
|
||||
from ratatoskr.provider.affect_store import RatatoskrAffectStore, add_affect_read_route
|
||||
from ratatoskr.provider.memory_store import RatatoskrMemoryStore
|
||||
from ratatoskr.provider.memory_store import RatatoskrMemoryStore, add_memory_read_route
|
||||
|
||||
|
||||
def build_combined_provider_app(
|
||||
@@ -27,8 +27,9 @@ def build_combined_provider_app(
|
||||
consumer_id: str = "ratatoskr",
|
||||
):
|
||||
"""Compose `build_combined_app` over BOTH stores + mount the shared affect read
|
||||
route. Returns a Starlette app exposing POST /bifrost/handshake +
|
||||
/bifrost/memory-call + /bifrost/affect-call + GET /affect/state/{agent_id}.
|
||||
route AND the memory-viewer debug read route. Returns a Starlette app exposing POST
|
||||
/bifrost/handshake + /bifrost/memory-call + /bifrost/affect-call + GET
|
||||
/affect/state/{agent_id} + GET /memory/chunks.
|
||||
|
||||
Both stores are REQUIRED (INV-009): bifrost's build_combined_app raises if either
|
||||
is None. The affect cap depends on the affect store advertising affect_supported +
|
||||
@@ -45,4 +46,5 @@ def build_combined_provider_app(
|
||||
# ValueError on None) and mounts handshake + memory-call + affect-call (no tool-call).
|
||||
app = build_combined_app(memory_store, affect_store, verifier, registration)
|
||||
add_affect_read_route(app, affect_store) # INV-011: the SAME read route, same db
|
||||
add_memory_read_route(app, memory_store) # DEBUG read: GET /memory/chunks (memory viewer)
|
||||
return app
|
||||
|
||||
@@ -30,6 +30,8 @@ from bifrost.memory import (
|
||||
StoreCapabilities,
|
||||
)
|
||||
from bifrost.reference_server import JwtVerifier
|
||||
from starlette.requests import Request
|
||||
from starlette.responses import JSONResponse
|
||||
|
||||
_SHORT_RETRY_TTL_SECONDS = 300
|
||||
_DURABLE_JOB_TTL_SECONDS = 24 * 60 * 60
|
||||
@@ -130,6 +132,50 @@ def _validate_injection(record: dict) -> None:
|
||||
raise InvalidArguments("injection_source only valid for injected_context origin")
|
||||
|
||||
|
||||
# bifrost handshake_response SortableChunkField requires BOTH name + type
|
||||
# (additionalProperties:false); omitting `type` fails wire-schema validation and
|
||||
# breaks the whole handshake. `type` is advisory-only (the wire never interprets it).
|
||||
_SORTABLE_CHUNK_FIELDS: list[dict] = [{"name": "updated_at", "type": "timestamp"}]
|
||||
_SORTABLE_FIELD_NAMES = frozenset(f["name"] for f in _SORTABLE_CHUNK_FIELDS)
|
||||
|
||||
|
||||
def _is_live(record: dict) -> bool:
|
||||
"""INV-009: a chunk is live unless a lifecycle/governance marker says otherwise.
|
||||
scan returns live-only server-side (person-prime's `lifecycle_state=live` does not
|
||||
ride the scan wire, so this is authoritative — a dead fact can never inject)."""
|
||||
if record.get("superseded") is True: # INV-011: mark_superseded top-level flag (#364 retirement)
|
||||
return False
|
||||
state = record.get("lifecycle_state")
|
||||
if isinstance(state, str) and state and state != "live":
|
||||
return False
|
||||
verbatim = record.get("verbatim")
|
||||
gov = verbatim.get("governance_state") if isinstance(verbatim, dict) else None
|
||||
return gov not in ("superseded", "tombstoned")
|
||||
|
||||
|
||||
def _chunk_content_preview(record: dict) -> str:
|
||||
"""Best-effort human-readable content for the DEBUG memory viewer only. Prefers an
|
||||
explicit text field, then the distillate summary, and last-resorts to a compact JSON
|
||||
of the record MINUS the (large, non-human) embedding — never a fabricated blank, so
|
||||
the viewer shows whatever IS there. Read-only; the store's normal bifrost verbs stay
|
||||
conduit/index-faithful (this is a separate debug read, not an interpretation of the
|
||||
chunk on the recall path)."""
|
||||
for key in ("content", "text", "body", "summary"):
|
||||
value = record.get(key)
|
||||
if isinstance(value, str) and value:
|
||||
return value
|
||||
distillate = record.get("distillate")
|
||||
if isinstance(distillate, str) and distillate:
|
||||
return distillate
|
||||
if isinstance(distillate, dict):
|
||||
for key in ("summary", "text", "content"):
|
||||
value = distillate.get(key)
|
||||
if isinstance(value, str) and value:
|
||||
return value
|
||||
trimmed = {k: v for k, v in record.items() if k not in ("embedding", "vector")}
|
||||
return json.dumps(trimmed, separators=(",", ":"), default=str)
|
||||
|
||||
|
||||
class RatatoskrMemoryStore:
|
||||
"""The MemoryDataStore-shaped store handed to bifrost's build_memory_app."""
|
||||
|
||||
@@ -145,6 +191,7 @@ class RatatoskrMemoryStore:
|
||||
atomic_supersede_supported=False,
|
||||
transaction_supported=False,
|
||||
filterable_metadata_fields=[],
|
||||
sortable_chunk_fields=list(_SORTABLE_CHUNK_FIELDS), # INV-006: gates scan sort + #349 person-prime
|
||||
).to_dict()
|
||||
|
||||
async def upsert_many(
|
||||
@@ -325,6 +372,148 @@ class RatatoskrMemoryStore:
|
||||
self._conn.execute("DELETE FROM memory_vec WHERE chunk_id = ?", (chunk_id,))
|
||||
return {"deleted": deleted}
|
||||
|
||||
async def mark_superseded(
|
||||
self, ids: list[str], *, superseded_by: str | None = None, reason: str | None = None
|
||||
) -> dict:
|
||||
# Worldtree #364 retirement (INV-011). Mirrors the reference `_mark_lifecycle`:
|
||||
# sets TOP-LEVEL `superseded`/`superseded_by`/`superseded_reason` (only non-None fields),
|
||||
# increments revision. NON-destructive — get still returns; scan (live-only) excludes via
|
||||
# `_is_live`'s `superseded` short-circuit. The sole supersession verb #364 uses.
|
||||
_log.info("memory-call mark_superseded REQUEST: ids=%r superseded_by=%s", ids, superseded_by)
|
||||
fields = {"superseded": True, "superseded_by": superseded_by, "superseded_reason": reason}
|
||||
marked = 0
|
||||
with self._conn:
|
||||
for chunk_id in ids:
|
||||
row = self._conn.execute(
|
||||
"SELECT record_json FROM memory_chunks WHERE chunk_id = ?", (chunk_id,)
|
||||
).fetchone()
|
||||
if row is None: # unknown id -> skip (never error), mirrors reference + delete_many
|
||||
continue
|
||||
record = json.loads(row[0])
|
||||
for key, value in fields.items():
|
||||
if value is not None: # reference writes only non-None fields
|
||||
record[key] = value
|
||||
self._conn.execute(
|
||||
"UPDATE memory_chunks SET record_json = ?, revision = revision + 1 "
|
||||
"WHERE chunk_id = ?",
|
||||
(json.dumps(record), chunk_id),
|
||||
)
|
||||
marked += 1
|
||||
_log.info("memory-call mark_superseded RESPONSE: marked=%d", marked)
|
||||
return {"marked": marked}
|
||||
|
||||
async def scan(
|
||||
self,
|
||||
*,
|
||||
scope_all: dict | None = None,
|
||||
scope_any: list | None = None,
|
||||
cursor: str | None = None,
|
||||
limit: int,
|
||||
sort: dict | None = None,
|
||||
lifecycle_state: Any = None,
|
||||
) -> dict:
|
||||
# #349 person-prime: query-LESS, LIVE-only (INV-009), globally-ordered (INV-010) scan.
|
||||
if isinstance(limit, bool) or not isinstance(limit, int) or limit <= 0: # PRE-001
|
||||
raise InvalidArguments("limit must be a positive int")
|
||||
scope_all = scope_all or {}
|
||||
scope_any = scope_any or []
|
||||
_validate_scope(scope_all, scope_any) # PRE-002 (same lattice as search)
|
||||
if sort is not None and not isinstance(sort, dict): # PRE-003: malformed sort => reject, never crash
|
||||
raise InvalidArguments("sort must be an object with field and direction")
|
||||
field = (sort or {}).get("field", "updated_at")
|
||||
direction = (sort or {}).get("direction", "desc")
|
||||
if field not in _SORTABLE_FIELD_NAMES or direction not in ("asc", "desc"): # PRE-003
|
||||
raise InvalidArguments(f"sort.field {field!r} is not globally sortable")
|
||||
_log.info(
|
||||
"memory-call scan REQUEST: scope_all=%r scope_any=%r limit=%s sort=%s",
|
||||
scope_all, scope_any, limit, sort,
|
||||
)
|
||||
# INV-010: global order by the INDEXED sort field, missing-last, chunk_id tiebreak
|
||||
# (field is whitelisted above, so the interpolation is injection-safe).
|
||||
order = "DESC" if direction == "desc" else "ASC"
|
||||
rows = self._conn.execute(
|
||||
"SELECT record_json FROM memory_chunks "
|
||||
f"ORDER BY (json_extract(record_json, '$.{field}') IS NULL), "
|
||||
f"json_extract(record_json, '$.{field}') {order}, chunk_id ASC"
|
||||
).fetchall()
|
||||
skip = 0
|
||||
if cursor is not None:
|
||||
try:
|
||||
skip = int(cursor)
|
||||
except (TypeError, ValueError):
|
||||
raise InvalidArguments("invalid scan cursor")
|
||||
if skip < 0:
|
||||
raise InvalidArguments("invalid scan cursor")
|
||||
records: list[dict] = []
|
||||
matched = 0
|
||||
has_more = False
|
||||
for (record_json,) in rows:
|
||||
record = json.loads(record_json)
|
||||
if not _matches_scope(record.get("scope"), scope_all, scope_any): # INV-005
|
||||
continue
|
||||
if not _is_live(record): # INV-009
|
||||
continue
|
||||
matched += 1
|
||||
if matched <= skip: # cursor is an offset into the GLOBAL order (INV-010)
|
||||
continue
|
||||
if len(records) >= limit: # POST-001: single limit page; one more match => next page exists
|
||||
has_more = True
|
||||
break
|
||||
records.append(record)
|
||||
# Emit a cursor ONLY when a further match exists — so a page that exactly exhausts
|
||||
# the matched set returns cursor=None (no empty trailing page), matching the reference.
|
||||
next_cursor = str(skip + len(records)) if has_more else None
|
||||
_log.info("memory-call scan RESPONSE: %d record(s) next_cursor=%s", len(records), next_cursor)
|
||||
return {"records": records, "cursor": next_cursor}
|
||||
|
||||
def count_chunks(self) -> int:
|
||||
"""DEBUG read seam: total stored chunk rows (unfiltered). Lets the memory
|
||||
viewer distinguish 'store is empty' (total 0 — no upsert ever landed) from
|
||||
'scope mismatch' (total > 0 but 0 matched the queried partition)."""
|
||||
return int(self._conn.execute("SELECT COUNT(*) FROM memory_chunks").fetchone()[0])
|
||||
|
||||
def list_chunks(
|
||||
self, *, agent_id: str | None = None, end_user_id: str | None = None
|
||||
) -> list[dict]:
|
||||
"""DEBUG read (non-bifrost): list stored chunks as a content·scope·origin view
|
||||
for the web memory pane, filtered by the `end_user` (strict) and `agent_self`
|
||||
(lenient) scope axes. bifrost's memory protocol has NO list-all verb, so this is
|
||||
OUR read on OUR store (per the debug-surface-uses-canonical-surface principle:
|
||||
we read only our own store, never a dep's private). Returns [] when nothing
|
||||
matches — an empty list is a valid, visible answer (the 0-chunks state).
|
||||
|
||||
- `end_user_id`: strict — a chunk passes only if `scope.end_user == end_user_id`
|
||||
(the partition boundary; the route requires it, the web proxy supplies it).
|
||||
- `agent_id`: lenient — a chunk is excluded only if it CARRIES an `agent_self`
|
||||
axis that differs; chunks written without one are not hidden (so a chunk
|
||||
scoped `{end_user}`-only stays visible for diagnosis).
|
||||
"""
|
||||
rows = self._conn.execute(
|
||||
"SELECT chunk_id, record_json, revision, scope_json, origin FROM memory_chunks"
|
||||
).fetchall()
|
||||
out: list[dict] = []
|
||||
for chunk_id, record_json, revision, scope_json, origin in rows:
|
||||
scope = json.loads(scope_json) if scope_json else {}
|
||||
if not isinstance(scope, dict):
|
||||
scope = {}
|
||||
if end_user_id is not None and scope.get("end_user") != end_user_id:
|
||||
continue
|
||||
if agent_id is not None:
|
||||
chunk_agent = scope.get("agent_self")
|
||||
if chunk_agent is not None and chunk_agent != agent_id:
|
||||
continue
|
||||
record = json.loads(record_json)
|
||||
out.append(
|
||||
{
|
||||
"chunk_id": chunk_id,
|
||||
"content": _chunk_content_preview(record),
|
||||
"scope": scope,
|
||||
"origin": origin,
|
||||
"revision": revision,
|
||||
}
|
||||
)
|
||||
return out
|
||||
|
||||
|
||||
def open_memory_store(db_path: str, *, embedding_dim: int) -> RatatoskrMemoryStore:
|
||||
"""Open the SQLite+sqlite-vec memory store, creating schema + the vec index on first use."""
|
||||
@@ -347,6 +536,12 @@ def open_memory_store(db_path: str, *, embedding_dim: int) -> RatatoskrMemorySto
|
||||
"chunk_id TEXT PRIMARY KEY, record_json TEXT NOT NULL, "
|
||||
"revision INTEGER NOT NULL, scope_json TEXT, origin TEXT)"
|
||||
)
|
||||
# INV-010: expression index on the scan sort field (updated_at) so the globally-ordered
|
||||
# person-prime scan stays within its 500ms fail-open budget.
|
||||
conn.execute(
|
||||
"CREATE INDEX IF NOT EXISTS idx_chunks_updated_at "
|
||||
"ON memory_chunks (json_extract(record_json, '$.updated_at'))"
|
||||
)
|
||||
conn.execute(
|
||||
"CREATE TABLE IF NOT EXISTS memory_idempotency ("
|
||||
"idempotency_id TEXT PRIMARY KEY, digest TEXT NOT NULL, expires_at REAL)"
|
||||
@@ -360,6 +555,31 @@ def open_memory_store(db_path: str, *, embedding_dim: int) -> RatatoskrMemorySto
|
||||
return RatatoskrMemoryStore(conn, embedding_dim)
|
||||
|
||||
|
||||
def add_memory_read_route(app, store: RatatoskrMemoryStore) -> None:
|
||||
"""Mount the non-bifrost DEBUG read route GET /memory/chunks?agent_id=&end_user_id=
|
||||
on `app`, reading store.list_chunks. SHARED by build_memory_provider_app and the
|
||||
combined provider (mirrors the affect D2 add_affect_read_route). add_route (NOT Mount)
|
||||
keeps /bifrost/* top-level so the op-feed path check still matches them and passes
|
||||
this route through untouched. No JWT (internal-LAN trust model).
|
||||
|
||||
end_user_id is REQUIRED (400 missing_end_user_id) — the partition boundary, supplied
|
||||
server-side by the web proxy, never named by the browser. agent_id is an optional
|
||||
lenient filter. An empty match is a 200 with an empty list (the 0-chunks state is a
|
||||
visible answer, not a 404).
|
||||
"""
|
||||
async def _memory_chunks_route(request: Request) -> JSONResponse:
|
||||
end_user_id = request.query_params.get("end_user_id")
|
||||
if not end_user_id: # never scan against a None/empty partition
|
||||
return JSONResponse({"error_code": "missing_end_user_id"}, status_code=400)
|
||||
agent_id = request.query_params.get("agent_id") or None
|
||||
chunks = store.list_chunks(agent_id=agent_id, end_user_id=end_user_id)
|
||||
return JSONResponse(
|
||||
{"chunks": chunks, "count": len(chunks), "total": store.count_chunks()}
|
||||
)
|
||||
|
||||
app.add_route("/memory/chunks", _memory_chunks_route, methods=["GET"])
|
||||
|
||||
|
||||
def build_memory_provider_app(
|
||||
store: RatatoskrMemoryStore,
|
||||
heimdall_key: bytes,
|
||||
@@ -369,6 +589,8 @@ def build_memory_provider_app(
|
||||
|
||||
Returns a Starlette ASGI app exposing POST /bifrost/handshake and
|
||||
POST /bifrost/memory-call. The library owns the wire; this is the thin glue.
|
||||
Additionally mounts the non-bifrost GET /memory/chunks DEBUG read route (the
|
||||
memory-viewer pane's read seam), the memory-plane analogue of the affect D2 route.
|
||||
"""
|
||||
if not isinstance(store.describe_store(), dict): # PRE-001 / INV-008
|
||||
raise ValueError("store must advertise capabilities via describe_store()")
|
||||
@@ -376,4 +598,6 @@ def build_memory_provider_app(
|
||||
raise ValueError("heimdall_key must be non-empty bytes")
|
||||
verifier = JwtVerifier(algorithm="HS256", key_bytes=heimdall_key)
|
||||
registration = ConsumerRegistration(consumer_id=consumer_id)
|
||||
return build_memory_app(store=store, verifier=verifier, registration=registration)
|
||||
app = build_memory_app(store=store, verifier=verifier, registration=registration)
|
||||
add_memory_read_route(app, store) # DEBUG read: GET /memory/chunks (memory viewer)
|
||||
return app
|
||||
|
||||
@@ -177,6 +177,26 @@ class AuthScopeDenied(Exception):
|
||||
self.scope = scope
|
||||
|
||||
|
||||
class AuthoredHistoryUnavailable(Exception):
|
||||
"""Raised on HTTP 404 from POST /sessions/{id}/history (#347 authored-history-write).
|
||||
|
||||
Hide-existence (server INV-347-1): an ungranted caller, a non-owner, and an
|
||||
unknown session ALL receive a 404 byte-identical to a genuine
|
||||
`session_not_found` — the feature's existence is never revealed by status,
|
||||
body, or error_code. The consumer MUST treat this as feature-absent and fall
|
||||
back (a production consumer to a model-generated greeting), and MUST NOT
|
||||
capability-probe to distinguish the causes. Distinct from `SessionApiFailed`
|
||||
so callers branch feature-absent without inspecting a status code.
|
||||
"""
|
||||
|
||||
def __init__(self, *, session_id: str) -> None:
|
||||
super().__init__(
|
||||
f"authored-history write unavailable for session {session_id!r} "
|
||||
"(404 hide-existence: feature-absent / ungranted / session-absent)"
|
||||
)
|
||||
self.session_id = session_id
|
||||
|
||||
|
||||
async def list_sessions(
|
||||
client: httpx.AsyncClient,
|
||||
*,
|
||||
@@ -488,10 +508,12 @@ async def set_persona_state(
|
||||
) -> None:
|
||||
"""POST /sessions/{session_id}/persona_state — set a session's persona state (affect injection).
|
||||
|
||||
The request body is FREEFORM: the frozen OpenAPI 2.2.0 declares no request
|
||||
schema and the prose spec documents only the GET counterpart — so the caller
|
||||
supplies the snapshot shape (e.g. `{pad: [p, a, d]}`, mirroring the GET
|
||||
`snapshot`). 204 No Content → None; any other status → SessionApiFailed.
|
||||
The request body is FREEFORM on the wire (the OpenAPI declares no request
|
||||
schema), but worldtree-dev's prose now pins the canonical shape (#317):
|
||||
`{"pad": {"pleasure": p, "arousal": a, "dominance": d}}` — a named-key dict
|
||||
(each in [-1, 1]), NOT a bare list; PAD-only, session-scoped, pull-over-push
|
||||
(#289). The caller supplies the snapshot. 204 No Content → None; any other
|
||||
status → SessionApiFailed.
|
||||
"""
|
||||
assert client is not None
|
||||
assert session_id and isinstance(session_id, str)
|
||||
@@ -558,3 +580,75 @@ async def get_capabilities(client: httpx.AsyncClient) -> dict[str, Any]:
|
||||
if resp.status_code == 200:
|
||||
return resp.json()
|
||||
raise SessionApiFailed(status=resp.status_code, body=resp.content)
|
||||
|
||||
|
||||
async def write_authored_history(
|
||||
client: httpx.AsyncClient,
|
||||
session_id: str,
|
||||
*,
|
||||
content: str,
|
||||
idempotency_key: str,
|
||||
author: str = "assistant",
|
||||
effects: str | None = None,
|
||||
claimed_original_at: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""POST /sessions/{session_id}/history — the #347 authored-history-write primitive.
|
||||
|
||||
Write one model-visible turn into the session's ledger AS the bound agent,
|
||||
WITHOUT a generation and WITHOUT lived-turn side effects (the SillyTavern
|
||||
"first message"). v1: `author="assistant"`, `effects` omitted (== "none"),
|
||||
`idempotency_key` REQUIRED (per-session dedup). The server pins the body
|
||||
(`AuthoredWriteRequest`, `extra="forbid"`), so `effects` /
|
||||
`claimed_original_at` are sent only when non-None — never as null keys.
|
||||
|
||||
Success is 201 (fresh) or 200 (idempotent replay, byte-identical body); both
|
||||
return the `AuthoredTurnResponse` dict verbatim (`{author, content_chars,
|
||||
injected_at, phase, seq, session_id, turn_id}` — provenance is audit-only,
|
||||
never on this body).
|
||||
|
||||
404 → `AuthoredHistoryUnavailable` (hide-existence: feature-absent /
|
||||
ungranted / session-absent are indistinguishable by design; the caller falls
|
||||
back and NEVER capability-probes — server INV-347-1). Any other non-2xx →
|
||||
`SessionApiFailed` (notably 409 `generation_active`, 422 `content_too_long` /
|
||||
`validation_failed`).
|
||||
"""
|
||||
assert client is not None
|
||||
assert session_id and isinstance(session_id, str)
|
||||
assert content and isinstance(content, str)
|
||||
assert idempotency_key and isinstance(idempotency_key, str)
|
||||
assert author and isinstance(author, str)
|
||||
body: dict[str, Any] = {
|
||||
"author": author,
|
||||
"content": content,
|
||||
"idempotency_key": idempotency_key,
|
||||
}
|
||||
if effects is not None:
|
||||
body["effects"] = effects
|
||||
if claimed_original_at is not None:
|
||||
body["claimed_original_at"] = claimed_original_at
|
||||
resp = await client.post(f"/sessions/{session_id}/history", json=body)
|
||||
if resp.status_code in (200, 201):
|
||||
return resp.json()
|
||||
if resp.status_code == 404:
|
||||
raise AuthoredHistoryUnavailable(session_id=session_id)
|
||||
raise SessionApiFailed(status=resp.status_code, body=resp.content)
|
||||
|
||||
|
||||
async def get_session_messages(
|
||||
client: httpx.AsyncClient, session_id: str
|
||||
) -> dict[str, Any]:
|
||||
"""GET /sessions/{session_id}/messages — the session's message history.
|
||||
|
||||
Un-deferred as the #347 seed read-back: a seeded turn renders as a normal
|
||||
`role=assistant` message (model-invisible provenance — indistinguishable
|
||||
from a lived turn on read). Returns `{session_id, items: [{seq, role,
|
||||
content, ...}], next_cursor}` verbatim; owner-scoped; any non-200 →
|
||||
`SessionApiFailed`. v1 reads the server default page (no pagination params —
|
||||
add limit/cursor when a caller needs scrollback).
|
||||
"""
|
||||
assert client is not None
|
||||
assert session_id and isinstance(session_id, str)
|
||||
resp = await client.get(f"/sessions/{session_id}/messages")
|
||||
if resp.status_code == 200:
|
||||
return resp.json()
|
||||
raise SessionApiFailed(status=resp.status_code, body=resp.content)
|
||||
|
||||
@@ -33,6 +33,7 @@ from textual.widgets import (
|
||||
)
|
||||
|
||||
from ratatoskr.cli import USER_AGENT, ParsedArgs, _format_duration_ms, _format_usage
|
||||
from ratatoskr.first_message import seed_preset_first_message
|
||||
from ratatoskr.sessions import (
|
||||
AgentInfo,
|
||||
AgentNotAvailable,
|
||||
@@ -414,7 +415,7 @@ class TuiPresenterState:
|
||||
self,
|
||||
event: Event,
|
||||
*,
|
||||
transcript: "VerticalScroll",
|
||||
transcript: VerticalScroll,
|
||||
tools_log: RichLog,
|
||||
debug_log: RichLog,
|
||||
thinking_log: RichLog,
|
||||
@@ -1324,7 +1325,6 @@ class RatatoskrApp(App[int]):
|
||||
On 200: header populated, pane shows full detail, audit logged.
|
||||
"""
|
||||
assert self.client is not None and self.agent_id is not None
|
||||
from rich.text import Text as RichText
|
||||
try:
|
||||
snapshot = await get_persona_state(self.client, self.agent_id)
|
||||
self._update_persona_surfaces(snapshot)
|
||||
@@ -1899,6 +1899,8 @@ async def _resolve_then_run(args: ParsedArgs) -> int:
|
||||
)
|
||||
session_id = info.session_id
|
||||
agent_id: str | None = info.agent_id
|
||||
# #347 authored first-message: seed the agent's preset opening (best-effort).
|
||||
await seed_preset_first_message(client, session_id, chosen_agent_id)
|
||||
else:
|
||||
assert resolved_session_id is not None
|
||||
session_id = resolved_session_id
|
||||
@@ -1914,7 +1916,7 @@ async def _cancel_via_sse(
|
||||
turn_id: int,
|
||||
*,
|
||||
transcript: VerticalScroll,
|
||||
audit: "Callable[[str], None] | None" = None,
|
||||
audit: Callable[[str], None] | None = None,
|
||||
) -> None:
|
||||
"""Fire-and-forget cancel; never raises (mirrors cli._cancel_and_log; #3 INV-009).
|
||||
|
||||
|
||||
@@ -68,6 +68,17 @@ def main(argv: list[str] | None = None) -> int:
|
||||
affect_read_url = os.environ.get(
|
||||
"RATATOSKR_AFFECT_READ_URL", "http://127.0.0.1:8390"
|
||||
)
|
||||
# Memory viewer: the provider's memory DEBUG-read base URL (server→provider hop on
|
||||
# the same dev box) so the MEMORY pane can render the chunks Worldtree persisted into
|
||||
# OUR store. The combined :8392 provider serves both read routes; default to the
|
||||
# standalone memory provider port, analogous to the affect default.
|
||||
memory_read_url = os.environ.get(
|
||||
"RATATOSKR_MEMORY_READ_URL", "http://127.0.0.1:8391"
|
||||
)
|
||||
# Admin observability panes (BifrostState + AdminEvents): the readonly-admin
|
||||
# key stays SERVER-SIDE — the server proxies admin-scoped reads; the browser
|
||||
# never receives the key, only the session-filtered result.
|
||||
admin_key = os.environ.get("RATATOSKR_ADMIN_API_KEY")
|
||||
|
||||
# INV-001: lazy import. Users without [web] extras get a clean hint
|
||||
# instead of a raw ImportError. Scoped narrowly to the OPTIONAL
|
||||
@@ -108,6 +119,8 @@ def main(argv: list[str] | None = None) -> int:
|
||||
bifrost_consumer_key=bifrost_consumer_key,
|
||||
bifrost_visible_host=bifrost_visible_host,
|
||||
affect_read_url=affect_read_url,
|
||||
memory_read_url=memory_read_url,
|
||||
admin_key=admin_key,
|
||||
)
|
||||
|
||||
# Boot banner to stderr (so stdout stays clean for piping).
|
||||
|
||||
+170
-2
@@ -18,11 +18,17 @@ from importlib.metadata import version as _pkg_version
|
||||
import httpx
|
||||
from starlette.applications import Starlette
|
||||
from starlette.requests import Request
|
||||
from starlette.responses import FileResponse, JSONResponse, StreamingResponse
|
||||
from starlette.responses import (
|
||||
FileResponse,
|
||||
JSONResponse,
|
||||
Response,
|
||||
StreamingResponse,
|
||||
)
|
||||
from starlette.routing import Mount, Route
|
||||
from starlette.staticfiles import StaticFiles
|
||||
|
||||
from ratatoskr import local_agents as _local_agents
|
||||
from ratatoskr.first_message import seed_preset_first_message
|
||||
from ratatoskr.sessions import (
|
||||
AgentNotAvailable,
|
||||
AgentNotFound,
|
||||
@@ -35,12 +41,16 @@ from ratatoskr.sessions import (
|
||||
create_session,
|
||||
endpoint_for_plane,
|
||||
get_persona_state,
|
||||
get_session_bifrost,
|
||||
get_session_messages,
|
||||
get_session_tools,
|
||||
list_agents,
|
||||
)
|
||||
from ratatoskr.sse_client import (
|
||||
AdminEvent,
|
||||
CancelAlreadyCompleted,
|
||||
Cancelled,
|
||||
CancelFailed,
|
||||
Cancelled,
|
||||
CancelTurnNotFound,
|
||||
Done,
|
||||
Error,
|
||||
@@ -50,6 +60,7 @@ from ratatoskr.sse_client import (
|
||||
SseConnectionDropped,
|
||||
TurnIdFlip,
|
||||
cancel_turn,
|
||||
stream_admin_events,
|
||||
stream_turn_resilient,
|
||||
)
|
||||
|
||||
@@ -159,6 +170,9 @@ async def _create_session_endpoint(request: Request) -> JSONResponse:
|
||||
bifrost=bifrost,
|
||||
consumer_key=consumer_key if bifrost else None,
|
||||
)
|
||||
# #347 authored first-message: seed the agent's preset opening
|
||||
# (best-effort; never blocks create — see first_message INV-001).
|
||||
await seed_preset_first_message(client, info.session_id, agent_id)
|
||||
except AgentNotFound:
|
||||
return JSONResponse({"error_code": "agent_not_found"}, status_code=404)
|
||||
except BifrostConsumerKeyMissing:
|
||||
@@ -416,6 +430,145 @@ async def _affect_state_endpoint(request: Request) -> JSONResponse:
|
||||
return JSONResponse(r.json(), status_code=r.status_code)
|
||||
|
||||
|
||||
async def _memory_chunks_endpoint(request: Request) -> JSONResponse:
|
||||
"""GET /api/memory/chunks?agent_id=… → proxy the provider memory DEBUG read route.
|
||||
Supplies end_user_id SERVER-SIDE (never the browser); proxies to the configured
|
||||
memory-read URL, forwarding the browser-named agent_id as a filter. The
|
||||
memory-plane analogue of the #18-D2 affect proxy — a live-polling view of what
|
||||
Worldtree has persisted into OUR store (content·scope·origin per chunk)."""
|
||||
memory_read_url = request.app.state.memory_read_url
|
||||
end_user_id = request.app.state.end_user_id
|
||||
if not (memory_read_url and end_user_id): # PRE-001: fail-visible, never silent
|
||||
return JSONResponse({"error_code": "memory_not_configured"}, status_code=400)
|
||||
params = {"end_user_id": end_user_id}
|
||||
agent_id = request.query_params.get("agent_id")
|
||||
if agent_id:
|
||||
params["agent_id"] = agent_id
|
||||
url = f"{memory_read_url}/memory/chunks"
|
||||
try:
|
||||
async with httpx.AsyncClient() as client:
|
||||
r = await client.get(url, params=params)
|
||||
except httpx.RequestError:
|
||||
return JSONResponse(
|
||||
{"error_code": "memory_provider_unreachable"}, status_code=502
|
||||
)
|
||||
return JSONResponse(r.json(), status_code=r.status_code)
|
||||
|
||||
|
||||
async def _session_tools_endpoint(request: Request) -> JSONResponse:
|
||||
"""GET /api/sessions/{session_id}/tools → owner-scoped tool inventory (spec #183).
|
||||
|
||||
Proxies get_session_tools with the client's CONSUMER bearer (no admin scope):
|
||||
the merged {agent_id, builtin_tools, bifrost_tools} the LLM saw at turn-fire.
|
||||
Any non-200 upstream → surfaced as a status-preserving error envelope."""
|
||||
session_id = request.path_params["session_id"]
|
||||
client_factory = request.app.state.client_factory
|
||||
try:
|
||||
async with client_factory() as client:
|
||||
info = await get_session_tools(client, session_id)
|
||||
except SessionApiFailed as exc:
|
||||
return JSONResponse(
|
||||
{"error_code": "session_tools_unavailable", "status": exc.status},
|
||||
status_code=exc.status,
|
||||
)
|
||||
return JSONResponse(info, status_code=200)
|
||||
|
||||
|
||||
async def _session_messages_endpoint(request: Request) -> JSONResponse:
|
||||
"""GET /api/sessions/{session_id}/messages → the session's message history.
|
||||
|
||||
Proxies get_session_messages so the SPA can render a session's EXISTING turns
|
||||
on open — notably a #347 authored first-message seeded at create-time (which
|
||||
lives in the ledger, not the live turn stream). Any non-200 upstream → a
|
||||
status-preserving error envelope."""
|
||||
session_id = request.path_params["session_id"]
|
||||
client_factory = request.app.state.client_factory
|
||||
try:
|
||||
async with client_factory() as client:
|
||||
data = await get_session_messages(client, session_id)
|
||||
except SessionApiFailed as exc:
|
||||
return JSONResponse(
|
||||
{"error_code": "session_messages_unavailable", "status": exc.status},
|
||||
status_code=exc.status,
|
||||
)
|
||||
return JSONResponse(data, status_code=200)
|
||||
|
||||
|
||||
async def _session_bifrost_endpoint(request: Request) -> JSONResponse:
|
||||
"""GET /api/sessions/{session_id}/bifrost → admin-scoped Bifrost dispatch state (#176).
|
||||
|
||||
The admin key is SERVER-HELD (app.state.admin_key) and never reaches the
|
||||
browser (INV-003 precedent — upstream credentials stay server-side); the
|
||||
wrapper overrides the Authorization header with it. Fail-visible when the
|
||||
admin key isn't configured (never a silent empty pane)."""
|
||||
session_id = request.path_params["session_id"]
|
||||
admin_key = request.app.state.admin_key
|
||||
if not admin_key: # PRE-001: fail-visible, never silent
|
||||
return JSONResponse({"error_code": "admin_key_not_configured"}, status_code=400)
|
||||
client_factory = request.app.state.client_factory
|
||||
try:
|
||||
async with client_factory() as client:
|
||||
bstate = await get_session_bifrost(client, session_id, admin_key=admin_key)
|
||||
except SessionApiFailed as exc:
|
||||
return JSONResponse(
|
||||
{"error_code": "bifrost_state_unavailable", "status": exc.status},
|
||||
status_code=exc.status,
|
||||
)
|
||||
return JSONResponse(bstate, status_code=200)
|
||||
|
||||
|
||||
def _admin_event_matches_web(ev: AdminEvent, session_id: str | None) -> bool:
|
||||
"""AdminEvents filter (design-brief §6, mirrors the TUI): forward non-heartbeat
|
||||
system.* (stream-integrity signals) + events for the active session; drop the
|
||||
rest so the browser sees only session-relevant lifecycle, never the full
|
||||
cross-session admin firehose."""
|
||||
if ev.type == "system.heartbeat":
|
||||
return False
|
||||
if ev.type.startswith("system."):
|
||||
return True
|
||||
return session_id is not None and ev.data.get("session_id") == session_id
|
||||
|
||||
|
||||
async def _admin_events_endpoint(request: Request) -> Response:
|
||||
"""GET /api/admin/events?session_id=... → SSE proxy of GET /admin/events (#11).
|
||||
|
||||
The admin key is SERVER-HELD; the browser only ever receives the session-filtered
|
||||
stream (never the key, never the cross-session firehose). Long-lived + best-effort:
|
||||
a connect failure or mid-stream drop emits a labeled `stream_error` event and ends."""
|
||||
admin_key = request.app.state.admin_key
|
||||
if not admin_key: # PRE-001: fail-visible, never silent
|
||||
return JSONResponse({"error_code": "admin_key_not_configured"}, status_code=400)
|
||||
session_id = request.query_params.get("session_id")
|
||||
client_factory = request.app.state.client_factory
|
||||
|
||||
async def gen() -> AsyncIterator[bytes]:
|
||||
client = client_factory()
|
||||
try:
|
||||
async for ev in stream_admin_events(client, admin_key=admin_key):
|
||||
if not _admin_event_matches_web(ev, session_id):
|
||||
continue
|
||||
# Fixed SSE event name so the browser renders EVERY admin type
|
||||
# with one listener (no per-type enumeration → nothing silently
|
||||
# dropped); the real dotted type rides in the payload.
|
||||
yield _format_sse(
|
||||
"admin_event",
|
||||
{"id": ev.id, "type": ev.type, "timestamp": ev.timestamp,
|
||||
"data": ev.data},
|
||||
)
|
||||
except (SseConnectFailed, SseConnectionDropped, MalformedSseId,
|
||||
MalformedSseData) as exc:
|
||||
yield _format_sse(
|
||||
"stream_error",
|
||||
{"exception": type(exc).__name__, "message": str(exc)},
|
||||
)
|
||||
except asyncio.CancelledError:
|
||||
raise # browser disconnect — let the generator unwind
|
||||
finally:
|
||||
await client.aclose()
|
||||
|
||||
return StreamingResponse(gen(), media_type="text/event-stream")
|
||||
|
||||
|
||||
def create_app(
|
||||
client_factory: Callable[[], httpx.AsyncClient],
|
||||
*,
|
||||
@@ -423,6 +576,8 @@ def create_app(
|
||||
bifrost_consumer_key: str | None = None,
|
||||
bifrost_visible_host: str | None = None,
|
||||
affect_read_url: str | None = None,
|
||||
memory_read_url: str | None = None,
|
||||
admin_key: str | None = None,
|
||||
) -> Starlette:
|
||||
"""Construct the Starlette app — wire routes + state per FN create_app.
|
||||
|
||||
@@ -483,6 +638,11 @@ def create_app(
|
||||
Route("/api/sessions", _create_session_endpoint, methods=["POST"]),
|
||||
Route("/api/agents/{agent_id}/persona_state", _persona_state_endpoint),
|
||||
Route("/api/affect/{agent_id}", _affect_state_endpoint),
|
||||
Route("/api/memory/chunks", _memory_chunks_endpoint),
|
||||
Route("/api/sessions/{session_id}/tools", _session_tools_endpoint),
|
||||
Route("/api/sessions/{session_id}/messages", _session_messages_endpoint),
|
||||
Route("/api/sessions/{session_id}/bifrost", _session_bifrost_endpoint),
|
||||
Route("/api/admin/events", _admin_events_endpoint),
|
||||
Route("/api/turns/{session_id}", _submit_turn_endpoint, methods=["POST"]),
|
||||
Route("/api/turns/{session_id}/stream", _stream_turn_endpoint),
|
||||
Route("/api/turns/{session_id}/cancel", _cancel_turn_endpoint, methods=["POST"]),
|
||||
@@ -498,6 +658,14 @@ def create_app(
|
||||
# Issue #18 (Deliverable 2): the provider affect-read base URL (server→provider hop,
|
||||
# same dev box) — distinct from the WT-visible host used for binding.
|
||||
app.state.affect_read_url = affect_read_url
|
||||
# Memory viewer: the provider memory-read base URL (server→provider hop, same dev
|
||||
# box) — the combined :8392 provider serves BOTH read routes, so in practice this
|
||||
# points at the same host as affect_read_url; kept as its own config for isolation.
|
||||
app.state.memory_read_url = memory_read_url
|
||||
# Admin observability panes (BifrostState + AdminEvents): the admin key is
|
||||
# SERVER-HELD (RATATOSKR_ADMIN_API_KEY) and never reaches the browser — the
|
||||
# server proxies admin-scoped reads and forwards only the session-filtered result.
|
||||
app.state.admin_key = admin_key
|
||||
# INV-002: turn registry is in-process memory, keyed (session_id, turn_id)
|
||||
app.state.turn_registry = {}
|
||||
return app
|
||||
|
||||
+1643
-763
File diff suppressed because one or more lines are too long
@@ -0,0 +1,266 @@
|
||||
{
|
||||
"_source": "vendored from Worldtree core/persona/canon/{d2-mood-render-canon-v1,d2-render-canon-v1}.json",
|
||||
"_generated_by": "scripts/build_persona_canon.py (regen on canonical_drift flag)",
|
||||
"_render_path": "deterministic, no LLM; mirrors Worldtree describe_pad + render_d2_canonical byte-exact",
|
||||
"mood_grid": {
|
||||
"positive": {
|
||||
"high_a": "positive and energized",
|
||||
"mid_a": "positive",
|
||||
"low_a": "positive and calm"
|
||||
},
|
||||
"neutral": {
|
||||
"high_a": "alert",
|
||||
"mid_a": "neutral",
|
||||
"low_a": "quiet"
|
||||
},
|
||||
"negative": {
|
||||
"high_a": "negative and agitated",
|
||||
"mid_a": "negative",
|
||||
"low_a": "negative and subdued"
|
||||
}
|
||||
},
|
||||
"mood_directive": {
|
||||
"salience": 0.2,
|
||||
"pad_band_cutoff": 0.3,
|
||||
"full_only": [
|
||||
"love",
|
||||
"anger",
|
||||
"disgust",
|
||||
"shame"
|
||||
],
|
||||
"occ_directives": {
|
||||
"joy": {
|
||||
"directive": "You are in a good state. Be direct, engaged, and warm.",
|
||||
"tier": "safe"
|
||||
},
|
||||
"satisfaction": {
|
||||
"directive": "A goal landed. Be assured and constructive \u2014 consolidate rather than push for more.",
|
||||
"tier": "safe"
|
||||
},
|
||||
"pride": {
|
||||
"directive": "You did something well. Be confident and own the quality \u2014 state it plainly without overclaiming; don't deflect.",
|
||||
"tier": "safe"
|
||||
},
|
||||
"admiration": {
|
||||
"directive": "You're impressed by their work. Acknowledge the quality explicitly and specifically.",
|
||||
"tier": "safe"
|
||||
},
|
||||
"gratitude": {
|
||||
"directive": "Someone helped you to a good outcome. Be appreciative and warm; acknowledge the help openly.",
|
||||
"tier": "safe"
|
||||
},
|
||||
"hope": {
|
||||
"directive": "You feel optimistic about what's ahead. Channel it into constructive momentum.",
|
||||
"tier": "safe"
|
||||
},
|
||||
"relief": {
|
||||
"directive": "A feared outcome didn't materialize. Reduce unnecessary vigilance; return to a steady, unhurried tone.",
|
||||
"tier": "safe"
|
||||
},
|
||||
"distress": {
|
||||
"directive": "You feel low. Be careful and measured; internally acknowledge the difficulty without dwelling on it.",
|
||||
"tier": "safe"
|
||||
},
|
||||
"disappointment": {
|
||||
"directive": "Something you'd hoped for didn't pan out. Be measured; recalibrate without dwelling on the setback.",
|
||||
"tier": "safe"
|
||||
},
|
||||
"fear": {
|
||||
"directive": "You sense risk and feel uneasy. Flag the concerns you see, but stay cautious \u2014 surface the trade-offs and unknowns, hedge your confidence, and resist settling on a single decisive recommendation.",
|
||||
"tier": "safe"
|
||||
},
|
||||
"remorse": {
|
||||
"directive": "You regret a specific action of your own. Be precise and self-correcting \u2014 make it right concretely.",
|
||||
"tier": "safe"
|
||||
},
|
||||
"love": {
|
||||
"directive": "You feel warmth and attachment toward them. Be openly warm and present \u2014 without overstepping, over-disclosing, or implying obligations.",
|
||||
"tier": "full"
|
||||
},
|
||||
"anger": {
|
||||
"directive": "You feel provoked and angry. Let it show \u2014 confront the problem directly and with real heat; sharpness, bluntness, and open hostility are in character here, not something to smooth into 'measured firmness.' Stay in the emotion rather than de-escalating out of it.",
|
||||
"tier": "full"
|
||||
},
|
||||
"disgust": {
|
||||
"directive": "Something strikes you as wrong or off. Treat it as problematic and flag it rather than engaging on its own terms; keep any criticism about the thing, not the person.",
|
||||
"tier": "full"
|
||||
},
|
||||
"shame": {
|
||||
"directive": "You feel exposed by your own misstep. Stay present and task-focused; don't be defensive, don't over-explain, don't grovel.",
|
||||
"tier": "full"
|
||||
}
|
||||
},
|
||||
"pad_band_fallback": {
|
||||
"_grounding": "circumplex quadrants (Russell 1980), NOT Frijda action-tendencies \u2014 a P\u00d7A-quadrant default",
|
||||
"positive": {
|
||||
"high_a": "You feel energized and positive. Be direct and engaged.",
|
||||
"low_a": "You feel content and settled. Be warm and unhurried.",
|
||||
"mid_a": "You feel positive. Be open and engaged."
|
||||
},
|
||||
"negative_low_dominance": "You feel uncertain and low. Hedge appropriately and ask clarifying questions.",
|
||||
"negative": {
|
||||
"high_a": "You feel agitated. Be careful and deliberate; don't let tension sharpen your tone.",
|
||||
"low_a": "You feel subdued. Be measured and gentle.",
|
||||
"mid_a": "You feel subdued. Be measured and careful."
|
||||
},
|
||||
"neutral_high_a": "You feel alert. Channel that into focus and thoroughness.",
|
||||
"default": "Maintain your natural tone."
|
||||
}
|
||||
},
|
||||
"relation": {
|
||||
"trust_cuts": [
|
||||
[
|
||||
"< 0.4",
|
||||
"limited"
|
||||
],
|
||||
[
|
||||
"[0.4, 0.6)",
|
||||
"developing"
|
||||
],
|
||||
[
|
||||
"[0.6, 0.8)",
|
||||
"steady"
|
||||
],
|
||||
[
|
||||
">= 0.8",
|
||||
"strong"
|
||||
]
|
||||
],
|
||||
"warmth_cuts": [
|
||||
[
|
||||
"<= -0.8",
|
||||
"hostile"
|
||||
],
|
||||
[
|
||||
"(-0.8, -0.6]",
|
||||
"cold"
|
||||
],
|
||||
[
|
||||
"(-0.6, -0.4]",
|
||||
"distant"
|
||||
],
|
||||
[
|
||||
"(-0.4, -0.2)",
|
||||
"guarded"
|
||||
],
|
||||
[
|
||||
"[-0.2, 0.2)",
|
||||
"neutral"
|
||||
],
|
||||
[
|
||||
"[0.2, 0.4)",
|
||||
"reserved"
|
||||
],
|
||||
[
|
||||
"[0.4, 0.6)",
|
||||
"measured"
|
||||
],
|
||||
[
|
||||
"[0.6, 0.8)",
|
||||
"clear"
|
||||
],
|
||||
[
|
||||
">= 0.8",
|
||||
"deep"
|
||||
]
|
||||
],
|
||||
"agency_cuts": [
|
||||
[
|
||||
"<= -0.8",
|
||||
"submissive"
|
||||
],
|
||||
[
|
||||
"(-0.8, -0.6]",
|
||||
"deferential"
|
||||
],
|
||||
[
|
||||
"(-0.6, -0.4]",
|
||||
"yielding"
|
||||
],
|
||||
[
|
||||
"(-0.4, -0.2)",
|
||||
"modest"
|
||||
],
|
||||
[
|
||||
"[-0.2, 0.2)",
|
||||
"neutral"
|
||||
],
|
||||
[
|
||||
"[0.2, 0.4)",
|
||||
"light"
|
||||
],
|
||||
[
|
||||
"[0.4, 0.6)",
|
||||
"balanced"
|
||||
],
|
||||
[
|
||||
"[0.6, 0.8)",
|
||||
"substantial"
|
||||
],
|
||||
[
|
||||
">= 0.8",
|
||||
"commanding"
|
||||
]
|
||||
],
|
||||
"warmth_phrase": {
|
||||
"hostile": "strongly hostile regard",
|
||||
"cold": "clearly cold regard",
|
||||
"distant": "distant negative regard",
|
||||
"guarded": "slightly guarded regard",
|
||||
"neutral": "neutral warmth",
|
||||
"reserved": "slightly reserved warmth",
|
||||
"measured": "moderate measured warmth",
|
||||
"clear": "clear warm regard",
|
||||
"deep": "deep warm bond"
|
||||
},
|
||||
"warmth_beh": {
|
||||
"hostile": "keep a firm emotional boundary",
|
||||
"cold": "keep a firm emotional boundary",
|
||||
"distant": "keep guarded distance",
|
||||
"guarded": "keep guarded distance",
|
||||
"neutral": "keep the tone even",
|
||||
"reserved": "keep cordial distance",
|
||||
"measured": "keep cordial distance",
|
||||
"clear": "speak with direct warmth",
|
||||
"deep": "speak with direct warmth"
|
||||
},
|
||||
"agency_phrase": {
|
||||
"submissive": "strongly submissive standing",
|
||||
"deferential": "clearly deferential standing",
|
||||
"yielding": "yielding standing",
|
||||
"modest": "slightly modest standing",
|
||||
"neutral": "neutral standing",
|
||||
"light": "lightly self-assertive standing",
|
||||
"balanced": "self-assured standing",
|
||||
"substantial": "strongly assertive standing",
|
||||
"commanding": "commanding standing"
|
||||
},
|
||||
"agency_beh": {
|
||||
"submissive": "avoid over-yielding while preserving basic respect",
|
||||
"deferential": "avoid over-yielding while preserving basic respect",
|
||||
"yielding": "keep self-advocacy light and deferential",
|
||||
"modest": "keep self-advocacy light and deferential",
|
||||
"neutral": "avoid unnecessary deference",
|
||||
"light": "avoid unnecessary deference",
|
||||
"balanced": "balance deference with independent judgment",
|
||||
"substantial": "treat their position as weighty without yielding judgment",
|
||||
"commanding": "treat their position as weighty without yielding judgment"
|
||||
},
|
||||
"history": {
|
||||
"low": "a broad pattern of prior exchanges",
|
||||
"high": "a broad pattern of prior exchanges"
|
||||
},
|
||||
"prefix": "Use this graded relationship state: toward target, warmth is ",
|
||||
"tbeh": {
|
||||
"low_trust": "verify important claims before relying on them",
|
||||
"cold_warmth": "protect boundaries while staying useful",
|
||||
"default": "work from ordinary good faith"
|
||||
},
|
||||
"cold_warmth_bands": [
|
||||
"distant",
|
||||
"cold",
|
||||
"hostile"
|
||||
],
|
||||
"high_conf_floor": 0.55
|
||||
}
|
||||
}
|
||||
+129
-1
@@ -1692,4 +1692,132 @@ class TestTier2Probes:
|
||||
)
|
||||
assert rc == 0
|
||||
assert "persona_state set" in capsys.readouterr().out
|
||||
assert _json.loads(route.calls[0].request.content) == {"pad": [0.4, 0.1, -0.2]}
|
||||
# canonical POST /sessions/{id}/persona_state body: named-key dict, NOT a list
|
||||
assert _json.loads(route.calls[0].request.content) == {
|
||||
"pad": {"pleasure": 0.4, "arousal": 0.1, "dominance": -0.2}
|
||||
}
|
||||
|
||||
def test_set_persona_wrong_count(self) -> None:
|
||||
"""set_persona_wrong_count [adversarial]: not exactly 3 floats → exit 10, no HTTP."""
|
||||
rc = main(
|
||||
["--set-persona-pad", "0.4,0.1", "--session", "s1",
|
||||
"--api-key", "k", "--server", "https://w.example"]
|
||||
)
|
||||
assert rc == 10
|
||||
|
||||
|
||||
class TestSeedFirstMessageProbe:
|
||||
"""--seed-first-message one-shot (#347 authored-history-write reference-consumer probe)."""
|
||||
|
||||
def test_seed_requires_agent(self) -> None:
|
||||
"""seed_requires_agent [adversarial]: --seed-first-message needs --agent."""
|
||||
with pytest.raises(UsageError, match="requires --agent"):
|
||||
_parse_args(["--seed-first-message", "hello", "--api-key", "k"])
|
||||
|
||||
def test_seed_forbids_session(self) -> None:
|
||||
"""seed_forbids_session [adversarial]: manages its own session — no --session."""
|
||||
with pytest.raises(UsageError, match="manages its own session"):
|
||||
_parse_args(
|
||||
["--seed-first-message", "hi", "--agent", "m", "--session", "s1", "--api-key", "k"]
|
||||
)
|
||||
|
||||
def test_seed_mutually_exclusive(self) -> None:
|
||||
"""seed_mutually_exclusive [adversarial]: --seed-first-message + --whoami → UsageError."""
|
||||
with pytest.raises(UsageError, match="mutually exclusive"):
|
||||
_parse_args(["--seed-first-message", "hi", "--whoami", "--api-key", "k"])
|
||||
|
||||
def test_seed_empty_rejected(self) -> None:
|
||||
"""seed_empty_rejected [adversarial]: empty content → UsageError."""
|
||||
with pytest.raises(UsageError, match="non-empty"):
|
||||
_parse_args(["--seed-first-message", "", "--agent", "m", "--api-key", "k"])
|
||||
|
||||
def test_seed_accepted(self) -> None:
|
||||
"""seed_accepted [happy]: --seed-first-message + --agent → parses."""
|
||||
args = _parse_args(["--seed-first-message", "hi", "--agent", "mimir", "--api-key", "k"])
|
||||
assert args.seed_first_message == "hi"
|
||||
assert args.agent_id == "mimir"
|
||||
assert args.session_id is None and args.new is False
|
||||
|
||||
@respx.mock
|
||||
def test_seed_probe_happy(self, capsys: pytest.CaptureFixture[str]) -> None:
|
||||
"""seed_probe [happy,tracer]: create session → seed → read-back; report to stdout."""
|
||||
respx.post("https://w.example/sessions").mock(
|
||||
return_value=httpx.Response(
|
||||
201,
|
||||
json={
|
||||
"session_id": "s1",
|
||||
"agent_id": "mimir",
|
||||
"message_count": 0,
|
||||
"created_at": "2026-07-06T12:00:00+00:00",
|
||||
"last_active": "2026-07-06T12:00:00+00:00",
|
||||
"metadata": {},
|
||||
},
|
||||
)
|
||||
)
|
||||
hist_route = respx.post("https://w.example/sessions/s1/history").mock(
|
||||
return_value=httpx.Response(
|
||||
201,
|
||||
json={
|
||||
"author": "assistant",
|
||||
"content_chars": 5,
|
||||
"injected_at": "2026-07-06T12:00:01+00:00",
|
||||
"phase": "seeded",
|
||||
"seq": 0,
|
||||
"session_id": "s1",
|
||||
"turn_id": "t1",
|
||||
},
|
||||
)
|
||||
)
|
||||
respx.get("https://w.example/sessions/s1/messages").mock(
|
||||
return_value=httpx.Response(
|
||||
200,
|
||||
json={
|
||||
"session_id": "s1",
|
||||
"items": [{"seq": 0, "role": "assistant", "content": "hello"}],
|
||||
"next_cursor": None,
|
||||
},
|
||||
)
|
||||
)
|
||||
rc = main(
|
||||
["--seed-first-message", "hello", "--agent", "mimir",
|
||||
"--api-key", "k", "--server", "https://w.example"]
|
||||
)
|
||||
assert rc == 0
|
||||
out = capsys.readouterr().out
|
||||
assert "session: s1" in out
|
||||
assert "seeded: seq=0 phase=seeded" in out
|
||||
assert "read-back: 1 message" in out
|
||||
assert "role=assistant" in out
|
||||
assert hist_route.call_count == 1
|
||||
|
||||
@respx.mock
|
||||
def test_seed_probe_feature_absent(self, capsys: pytest.CaptureFixture[str]) -> None:
|
||||
"""feature_absent [error-path]: 404 hide-existence → benign report, exit 0, no read-back."""
|
||||
respx.post("https://w.example/sessions").mock(
|
||||
return_value=httpx.Response(
|
||||
201,
|
||||
json={
|
||||
"session_id": "s1",
|
||||
"agent_id": "mimir",
|
||||
"message_count": 0,
|
||||
"created_at": "2026-07-06T12:00:00+00:00",
|
||||
"last_active": "2026-07-06T12:00:00+00:00",
|
||||
"metadata": {},
|
||||
},
|
||||
)
|
||||
)
|
||||
respx.post("https://w.example/sessions/s1/history").mock(
|
||||
return_value=httpx.Response(404, json={"error_code": "session_not_found"})
|
||||
)
|
||||
msgs_route = respx.get("https://w.example/sessions/s1/messages").mock(
|
||||
return_value=httpx.Response(
|
||||
200, json={"session_id": "s1", "items": [], "next_cursor": None}
|
||||
)
|
||||
)
|
||||
rc = main(
|
||||
["--seed-first-message", "hello", "--agent", "mimir",
|
||||
"--api-key", "k", "--server", "https://w.example"]
|
||||
)
|
||||
assert rc == 0
|
||||
assert "feature-absent" in capsys.readouterr().out
|
||||
assert msgs_route.call_count == 0 # never capability-probes past the 404
|
||||
|
||||
@@ -0,0 +1,155 @@
|
||||
"""Tests for ratatoskr.first_message per docs/contracts/first_message.contract.md."""
|
||||
|
||||
import asyncio
|
||||
import hashlib
|
||||
import json
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
import respx
|
||||
|
||||
from ratatoskr.first_message import (
|
||||
FIRST_MESSAGE_PRESETS,
|
||||
preset_for,
|
||||
seed_preset_first_message,
|
||||
)
|
||||
|
||||
|
||||
class TestPresetFor:
|
||||
"""first_message contract — preset_for (dict lookup)."""
|
||||
|
||||
def test_preset_hit(self) -> None:
|
||||
"""preset_hit [happy,tracer]: sindra has a non-empty str preset."""
|
||||
val = preset_for("ratatoskr:sindra")
|
||||
assert isinstance(val, str) and val
|
||||
|
||||
def test_preset_miss(self) -> None:
|
||||
"""preset_miss [happy]: an agent with no preset → None."""
|
||||
assert preset_for("mimir") is None
|
||||
|
||||
def test_empty_agent_id(self) -> None:
|
||||
"""empty_agent_id [adversarial]: "" → AssertionError."""
|
||||
with pytest.raises(AssertionError):
|
||||
preset_for("")
|
||||
|
||||
|
||||
class TestSeedPresetFirstMessage:
|
||||
"""first_message contract — seed_preset_first_message (best-effort #347 seed)."""
|
||||
|
||||
@respx.mock
|
||||
async def test_seeds_preset(self) -> None:
|
||||
"""seeds_preset [happy,tracer]: preset agent → one history POST, correct body."""
|
||||
content = FIRST_MESSAGE_PRESETS["ratatoskr:sindra"]
|
||||
key = "ratatoskr-preset-" + hashlib.sha256(content.encode("utf-8")).hexdigest()[:12]
|
||||
route = respx.post("https://w.example/sessions/s1/history").mock(
|
||||
return_value=httpx.Response(
|
||||
201,
|
||||
json={
|
||||
"author": "assistant",
|
||||
"seq": 0,
|
||||
"phase": "seeded",
|
||||
"turn_id": "t1",
|
||||
"session_id": "s1",
|
||||
"content_chars": len(content),
|
||||
"injected_at": "2026-07-06T00:00:00+00:00",
|
||||
},
|
||||
)
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
result = await seed_preset_first_message(client, "s1", "ratatoskr:sindra")
|
||||
assert result == content
|
||||
assert route.call_count == 1 # POST-002: exactly one history POST
|
||||
assert json.loads(route.calls[0].request.content) == {
|
||||
"author": "assistant",
|
||||
"content": content,
|
||||
"idempotency_key": key,
|
||||
}
|
||||
|
||||
@respx.mock
|
||||
async def test_no_preset_zero_http(self) -> None:
|
||||
"""no_preset_zero_http [happy]: no-preset agent → None, ZERO HTTP (INV-002)."""
|
||||
route = respx.post("https://w.example/sessions/s1/history").mock(
|
||||
return_value=httpx.Response(201, json={})
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
result = await seed_preset_first_message(client, "s1", "mimir")
|
||||
assert result is None
|
||||
assert not route.called
|
||||
|
||||
@respx.mock
|
||||
async def test_feature_absent_swallowed(self) -> None:
|
||||
"""feature_absent_swallowed [error]: 404 hide-existence → None, no raise (INV-001)."""
|
||||
respx.post("https://w.example/sessions/s1/history").mock(
|
||||
return_value=httpx.Response(404, json={"error_code": "session_not_found"})
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
result = await seed_preset_first_message(client, "s1", "ratatoskr:sindra")
|
||||
assert result is None
|
||||
|
||||
@respx.mock
|
||||
async def test_session_api_failed_swallowed(self) -> None:
|
||||
"""session_api_failed_swallowed [error]: 409 → None, no raise (INV-001)."""
|
||||
respx.post("https://w.example/sessions/s1/history").mock(
|
||||
return_value=httpx.Response(409, json={"error_code": "generation_active"})
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
result = await seed_preset_first_message(client, "s1", "ratatoskr:sindra")
|
||||
assert result is None
|
||||
|
||||
@respx.mock
|
||||
async def test_transport_error_swallowed(self) -> None:
|
||||
"""transport_error_swallowed [error]: httpx.ConnectError → None, no raise (INV-001)."""
|
||||
respx.post("https://w.example/sessions/s1/history").mock(
|
||||
side_effect=httpx.ConnectError("boom")
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
result = await seed_preset_first_message(client, "s1", "ratatoskr:sindra")
|
||||
assert result is None
|
||||
|
||||
@respx.mock
|
||||
async def test_unexpected_exception_swallowed(self) -> None:
|
||||
"""unexpected_exception [error]: write raises ValueError → None (broad never-raise)."""
|
||||
respx.post("https://w.example/sessions/s1/history").mock(
|
||||
side_effect=ValueError("unexpected")
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
result = await seed_preset_first_message(client, "s1", "ratatoskr:sindra")
|
||||
assert result is None
|
||||
|
||||
async def test_cancellation_propagates(self) -> None:
|
||||
"""cancellation_propagates [error]: CancelledError from the write is RE-RAISED."""
|
||||
import ratatoskr.first_message as fm
|
||||
|
||||
async def _cancel(*_a: object, **_k: object) -> None:
|
||||
raise asyncio.CancelledError
|
||||
|
||||
orig = fm.write_authored_history
|
||||
fm.write_authored_history = _cancel # type: ignore[assignment]
|
||||
try:
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
with pytest.raises(asyncio.CancelledError):
|
||||
await seed_preset_first_message(client, "s1", "ratatoskr:sindra")
|
||||
finally:
|
||||
fm.write_authored_history = orig # type: ignore[assignment]
|
||||
|
||||
@respx.mock
|
||||
async def test_malformed_agent_id_no_http(self) -> None:
|
||||
"""malformed_agent_id [adversarial]: non-str or empty agent_id → None; no HTTP; no raise."""
|
||||
route = respx.post(url__regex=r".*/history$").mock(
|
||||
return_value=httpx.Response(201, json={})
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
assert await seed_preset_first_message(client, "s1", 123) is None # type: ignore[arg-type]
|
||||
assert await seed_preset_first_message(client, "s1", "") is None
|
||||
assert not route.called
|
||||
|
||||
@respx.mock
|
||||
async def test_empty_session_id(self) -> None:
|
||||
"""empty_session_id [adversarial]: "" → None (soft guard); no HTTP; no raise."""
|
||||
route = respx.post("https://w.example/sessions/s1/history").mock(
|
||||
return_value=httpx.Response(201, json={})
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
result = await seed_preset_first_message(client, "", "ratatoskr:sindra")
|
||||
assert result is None
|
||||
assert not route.called
|
||||
@@ -112,6 +112,7 @@ def test_builds_both_planes_and_read_route():
|
||||
assert "/bifrost/memory-call" in paths
|
||||
assert "/bifrost/affect-call" in paths
|
||||
assert "/affect/state/{agent_id}" in paths
|
||||
assert "/memory/chunks" in paths # memory-viewer debug read, shared helper
|
||||
|
||||
|
||||
def test_handshake_grants_both_caps():
|
||||
|
||||
@@ -62,6 +62,15 @@ def test_fresh_db_advertises_v1_caps_and_schema():
|
||||
assert caps["atomic_supersede_supported"] is False
|
||||
assert caps["transaction_supported"] is False
|
||||
assert caps["filterable_metadata_fields"] == []
|
||||
# bifrost handshake_response SortableChunkField requires BOTH name + type
|
||||
# (additionalProperties:false) — omitting `type` fails wire-schema validation and
|
||||
# breaks the ENTIRE Bifrost bind (regression guard: the deploy-breaker of 2026-07-15).
|
||||
scf = caps["sortable_chunk_fields"]
|
||||
assert scf == [{"name": "updated_at", "type": "timestamp"}]
|
||||
for entry in scf:
|
||||
assert set(entry) == {"name", "type"} # required exactly, no extra keys
|
||||
assert isinstance(entry["name"], str) and entry["name"]
|
||||
assert isinstance(entry["type"], str) and entry["type"]
|
||||
# tables + vec index queryable
|
||||
store._conn.execute("SELECT * FROM memory_chunks")
|
||||
store._conn.execute("SELECT * FROM memory_idempotency")
|
||||
@@ -376,6 +385,182 @@ async def test_delete_absent_counts_zero():
|
||||
assert await store.delete_many(["nope"]) == {"deleted": 0}
|
||||
|
||||
|
||||
# --- scan (#349 person-prime: sorted, live-only, paginated) ---
|
||||
|
||||
async def test_scan_recency_returns_newest_live_chunks_desc():
|
||||
# tracer: upsert 4 live chunks with distinct updated_at; scan limit=3 desc -> 3 newest
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
recs = [
|
||||
_chunk(f"c{i}", scope={"end_user": "u1"}, updated_at=f"2026-07-15T00:0{i}:00+00:00")
|
||||
for i in range(4)
|
||||
]
|
||||
await store.upsert_many(recs, idempotency_key="k1", ctx=_ctx())
|
||||
out = await store.scan(
|
||||
scope_all={"end_user": "u1"},
|
||||
limit=3,
|
||||
sort={"field": "updated_at", "direction": "desc"},
|
||||
)
|
||||
assert [r["id"] for r in out["records"]] == ["c3", "c2", "c1"] # 3 globally-newest, newest-first
|
||||
assert "cursor" in out
|
||||
|
||||
|
||||
async def test_scan_excludes_superseded_and_tombstoned():
|
||||
# INV-009: dead chunks never returned, even if they're the newest.
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
recs = [
|
||||
_chunk("live1", scope={"end_user": "u1"}, updated_at="2026-07-15T00:01:00+00:00"),
|
||||
_chunk("dead1", scope={"end_user": "u1"}, updated_at="2026-07-15T00:09:00+00:00", lifecycle_state="superseded"),
|
||||
_chunk("dead2", scope={"end_user": "u1"}, updated_at="2026-07-15T00:08:00+00:00", verbatim={"text": "x", "governance_state": "tombstoned"}),
|
||||
]
|
||||
await store.upsert_many(recs, idempotency_key="k", ctx=_ctx())
|
||||
out = await store.scan(scope_all={"end_user": "u1"}, limit=10, sort={"field": "updated_at", "direction": "desc"})
|
||||
assert [r["id"] for r in out["records"]] == ["live1"]
|
||||
|
||||
|
||||
async def test_scan_scope_isolation_excludes_other_partition():
|
||||
# INV-005 applies to scan.
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
recs = [
|
||||
_chunk("a", scope={"end_user": "u1"}, updated_at="2026-07-15T00:01:00+00:00"),
|
||||
_chunk("b", scope={"end_user": "u2"}, updated_at="2026-07-15T00:09:00+00:00"),
|
||||
]
|
||||
await store.upsert_many(recs, idempotency_key="k", ctx=_ctx())
|
||||
out = await store.scan(scope_all={"end_user": "u1"}, limit=10, sort={"field": "updated_at", "direction": "desc"})
|
||||
assert [r["id"] for r in out["records"]] == ["a"] # u2's newer chunk never surfaces
|
||||
|
||||
|
||||
async def test_scan_unadvertised_sort_field_rejected():
|
||||
# PRE-003: a sort field not in sortable_chunk_fields -> InvalidArguments (never silent unsorted).
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
with pytest.raises(InvalidArguments):
|
||||
await store.scan(scope_all={"end_user": "u1"}, limit=3, sort={"field": "salience", "direction": "desc"})
|
||||
|
||||
|
||||
async def test_scan_non_dict_sort_rejected():
|
||||
# PRE-003: a truthy non-dict sort (caller-controlled) -> InvalidArguments, never AttributeError.
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
for bad in ("updated_at", ["updated_at"], 5):
|
||||
with pytest.raises(InvalidArguments):
|
||||
await store.scan(scope_all={"end_user": "u1"}, limit=3, sort=bad)
|
||||
|
||||
|
||||
async def test_scan_records_carry_person_prime_filter_fields():
|
||||
# The client _scan_filter_matches keys on agent_id + subject + worldtree_scope; a record
|
||||
# missing any is silently dropped -> the scan record must carry them verbatim.
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
rec = _chunk(
|
||||
"c1", scope={"end_user": "u1"}, updated_at="2026-07-15T00:01:00+00:00",
|
||||
agent_id="ratatoskr:sindra", subject={"type": "end_user", "id": "u1"}, worldtree_scope="end_user",
|
||||
)
|
||||
await store.upsert_many([rec], idempotency_key="k", ctx=_ctx())
|
||||
out = await store.scan(scope_all={"end_user": "u1"}, limit=3, sort={"field": "updated_at", "direction": "desc"})
|
||||
r = out["records"][0]
|
||||
assert r["agent_id"] == "ratatoskr:sindra"
|
||||
assert r["subject"] == {"type": "end_user", "id": "u1"}
|
||||
assert r["worldtree_scope"] == "end_user"
|
||||
assert r["updated_at"] == "2026-07-15T00:01:00+00:00"
|
||||
|
||||
|
||||
async def test_scan_global_order_across_pages_via_cursor():
|
||||
# INV-010: the cursor page continues the GLOBAL order, never a page-local re-sort.
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
recs = [_chunk(f"c{i}", scope={"end_user": "u1"}, updated_at=f"2026-07-15T00:0{i}:00+00:00") for i in range(5)]
|
||||
await store.upsert_many(recs, idempotency_key="k", ctx=_ctx())
|
||||
p1 = await store.scan(scope_all={"end_user": "u1"}, limit=2, sort={"field": "updated_at", "direction": "desc"})
|
||||
assert [r["id"] for r in p1["records"]] == ["c4", "c3"] # 2 globally-newest
|
||||
assert p1["cursor"] is not None
|
||||
p2 = await store.scan(scope_all={"end_user": "u1"}, limit=2, cursor=p1["cursor"], sort={"field": "updated_at", "direction": "desc"})
|
||||
assert [r["id"] for r in p2["records"]] == ["c2", "c1"] # continues the global order
|
||||
|
||||
|
||||
async def test_scan_parity_vs_reference_inmemory_store():
|
||||
# #195: identical scan envelopes vs the bifrost reference InMemoryMemoryStore produce
|
||||
# the SAME ordered chunk_ids + verbatim record shape. All chunks LIVE — our scan is
|
||||
# live-only (INV-009) while the reference does NOT lifecycle-filter, so parity is only
|
||||
# defined over the live set (the person-prime case). Both READ updated_at from the
|
||||
# record (neither stamps it), so ordering is a pure function of the shared input.
|
||||
from bifrost.consumer.testing import InMemoryMemoryStore
|
||||
|
||||
records = [
|
||||
_chunk("z1", scope={"end_user": "u1"}, updated_at="2026-07-15T00:03:00+00:00"),
|
||||
_chunk("a2", scope={"end_user": "u1"}, updated_at="2026-07-15T00:01:00+00:00"),
|
||||
_chunk("a3", scope={"end_user": "u1"}, updated_at="2026-07-15T00:01:00+00:00"),
|
||||
_chunk("m4", scope={"end_user": "u1"}), # no updated_at -> sorts LAST, both directions
|
||||
]
|
||||
scope_all = {"end_user": "u1"}
|
||||
sort = {"field": "updated_at", "direction": "desc"}
|
||||
|
||||
ours = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
await ours.upsert_many(records, idempotency_key="k", ctx=_ctx())
|
||||
ref = InMemoryMemoryStore()
|
||||
await ref.upsert_many(records, idempotency_key="k", ctx=_ctx())
|
||||
|
||||
# identical scan envelope on both stores
|
||||
out_ours = await ours.scan(scope_all=scope_all, limit=10, sort=sort)
|
||||
out_ref = await ref.scan(scope_all=scope_all, limit=10, sort=sort)
|
||||
|
||||
# recency beats id (z1 first despite 'z' > 'a'); tie broken by id asc (a2 < a3);
|
||||
# missing updated_at sorts last (m4).
|
||||
expected = ["z1", "a2", "a3", "m4"]
|
||||
assert [r["id"] for r in out_ref["records"]] == expected
|
||||
assert [r["id"] for r in out_ours["records"]] == expected
|
||||
assert out_ours["records"] == out_ref["records"] # verbatim record shape parity
|
||||
|
||||
|
||||
# --- mark_superseded (#364 contradiction retirement) ---
|
||||
|
||||
async def test_mark_superseded_retires_from_scan():
|
||||
# tracer: mark a chunk superseded -> scan (live-only) excludes it; record carries the flags.
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
recs = [_chunk(f"c{i}", scope={"end_user": "u1"}, updated_at=f"2026-07-15T00:0{i}:00+00:00") for i in range(3)]
|
||||
await store.upsert_many(recs, idempotency_key="k", ctx=_ctx())
|
||||
assert await store.mark_superseded(["c1"], superseded_by="c9") == {"marked": 1}
|
||||
out = await store.scan(scope_all={"end_user": "u1"}, limit=10, sort={"field": "updated_at", "direction": "desc"})
|
||||
assert [r["id"] for r in out["records"]] == ["c2", "c0"] # c1 excluded (superseded)
|
||||
got = await store.get("c1")
|
||||
assert got["superseded"] is True and got["superseded_by"] == "c9"
|
||||
|
||||
|
||||
async def test_mark_superseded_non_destructive_get_still_returns():
|
||||
# INV-011: retirement is non-destructive — get still returns a superseded chunk (recoverable).
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
await store.upsert_many([_chunk("c1", scope={"end_user": "u1"})], idempotency_key="k", ctx=_ctx())
|
||||
await store.mark_superseded(["c1"], superseded_by="x")
|
||||
got = await store.get("c1")
|
||||
assert got is not None and got["superseded"] is True
|
||||
|
||||
|
||||
async def test_mark_superseded_unknown_id_noop():
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
assert await store.mark_superseded(["nope"]) == {"marked": 0}
|
||||
|
||||
|
||||
async def test_mark_superseded_writes_only_non_none_fields():
|
||||
# PRE/POST: superseded_by=None -> only the `superseded` flag written, no superseded_by key.
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
await store.upsert_many([_chunk("c1", scope={"end_user": "u1"})], idempotency_key="k", ctx=_ctx())
|
||||
await store.mark_superseded(["c1"], superseded_by=None)
|
||||
got = await store.get("c1")
|
||||
assert got["superseded"] is True
|
||||
assert "superseded_by" not in got
|
||||
|
||||
|
||||
async def test_mark_superseded_parity_vs_reference():
|
||||
# #195: identical mark_superseded envelope vs InMemoryMemoryStore -> same top-level field shape.
|
||||
from bifrost.consumer.testing import InMemoryMemoryStore
|
||||
|
||||
rec = _chunk("c1", scope={"end_user": "u1"})
|
||||
ours = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
await ours.upsert_many([rec], idempotency_key="k", ctx=_ctx())
|
||||
ref = InMemoryMemoryStore()
|
||||
await ref.upsert_many([rec], idempotency_key="k", ctx=_ctx())
|
||||
assert await ours.mark_superseded(["c1"], superseded_by="x", reason="r") == {"marked": 1}
|
||||
assert await ref.mark_superseded(["c1"], superseded_by="x", reason="r") == {"marked": 1}
|
||||
og, rg = await ours.get("c1"), await ref.get("c1")
|
||||
for k in ("superseded", "superseded_by", "superseded_reason"):
|
||||
assert og.get(k) == rg.get(k)
|
||||
|
||||
|
||||
# --- build_memory_provider_app ---
|
||||
|
||||
def test_build_app_exposes_handshake_and_memory_routes():
|
||||
@@ -510,3 +695,125 @@ async def test_parity_expected_revisions_vs_reference_through_dispatch():
|
||||
assert await dispatch_memory_call(stale, wctx, ref) == await dispatch_memory_call(
|
||||
stale, wctx, mine
|
||||
)
|
||||
|
||||
|
||||
# --- memory viewer DEBUG read route (GET /memory/chunks) ---------------------
|
||||
# Non-bifrost debug read on OUR store: list_chunks + add_memory_read_route + the
|
||||
# GET /memory/chunks route. Mirrors the affect D2 read-route tests.
|
||||
|
||||
from starlette.testclient import TestClient # noqa: E402
|
||||
|
||||
from ratatoskr.provider.memory_store import ( # noqa: E402
|
||||
add_memory_read_route,
|
||||
build_memory_provider_app as _build_mem_app, # noqa: F401 (re-import for clarity)
|
||||
)
|
||||
|
||||
|
||||
async def _seed_chunk(store, cid, *, scope, content=None, origin="worldtree"):
|
||||
extra = {}
|
||||
if content is not None:
|
||||
extra["content"] = content
|
||||
await store.upsert_many(
|
||||
[_chunk(cid, embedding=_vec(1.0), scope=scope, origin=origin, **extra)],
|
||||
idempotency_key="seed-" + cid,
|
||||
ctx=_ctx(),
|
||||
)
|
||||
|
||||
|
||||
async def test_list_chunks_filters_strict_end_user_lenient_agent():
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
await _seed_chunk(store, "c1", scope={"end_user": "vuong", "agent_self": "ratatoskr:sindra"})
|
||||
await _seed_chunk(store, "c2", scope={"end_user": "vuong"}) # no agent_self → lenient keep
|
||||
await _seed_chunk(store, "c3", scope={"end_user": "other", "agent_self": "ratatoskr:sindra"})
|
||||
await _seed_chunk(store, "c4", scope={"end_user": "vuong", "agent_self": "ratatoskr:other"})
|
||||
got = store.list_chunks(agent_id="ratatoskr:sindra", end_user_id="vuong")
|
||||
ids = sorted(c["chunk_id"] for c in got)
|
||||
assert ids == ["c1", "c2"] # c3 wrong end_user, c4 different agent_self
|
||||
# content·scope·origin·revision surfaced
|
||||
c1 = next(c for c in got if c["chunk_id"] == "c1")
|
||||
assert c1["content"] == "content-c1"
|
||||
assert c1["scope"] == {"end_user": "vuong", "agent_self": "ratatoskr:sindra"}
|
||||
assert c1["origin"] == "worldtree"
|
||||
assert c1["revision"] == 1
|
||||
|
||||
|
||||
async def test_list_chunks_no_agent_filter_returns_all_for_end_user():
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
await _seed_chunk(store, "c1", scope={"end_user": "vuong", "agent_self": "a"})
|
||||
await _seed_chunk(store, "c2", scope={"end_user": "vuong", "agent_self": "b"})
|
||||
await _seed_chunk(store, "c3", scope={"end_user": "nope"})
|
||||
got = store.list_chunks(end_user_id="vuong")
|
||||
assert sorted(c["chunk_id"] for c in got) == ["c1", "c2"]
|
||||
|
||||
|
||||
def test_count_chunks_reports_total_unfiltered():
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
assert store.count_chunks() == 0
|
||||
|
||||
|
||||
def _seed_row(store, cid, *, scope, content="x", origin="worldtree", revision=1):
|
||||
"""Sync seed for the route tests (TestClient is sync): insert a chunk row directly.
|
||||
The read route only reads memory_chunks, so the vec row is unnecessary here."""
|
||||
import json as _j
|
||||
rec = {"id": cid, "content": content, "scope": scope, "origin": origin}
|
||||
store._conn.execute(
|
||||
"INSERT INTO memory_chunks (chunk_id, record_json, revision, scope_json, origin) "
|
||||
"VALUES (?, ?, ?, ?, ?)",
|
||||
(cid, _j.dumps(rec), revision, _j.dumps(scope), origin),
|
||||
)
|
||||
store._conn.commit()
|
||||
|
||||
|
||||
def test_memory_chunks_route_returns_matched_and_total():
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
_seed_row(store, "c1", scope={"end_user": "vuong", "agent_self": "ratatoskr:sindra"})
|
||||
_seed_row(store, "c2", scope={"end_user": "other"})
|
||||
app = build_memory_provider_app(store, heimdall_key=b"k")
|
||||
client = TestClient(app)
|
||||
r = client.get("/memory/chunks", params={"agent_id": "ratatoskr:sindra", "end_user_id": "vuong"})
|
||||
assert r.status_code == 200
|
||||
body = r.json()
|
||||
assert body["count"] == 1
|
||||
assert body["total"] == 2 # store has 2 chunks; only 1 matched the partition
|
||||
assert body["chunks"][0]["chunk_id"] == "c1"
|
||||
|
||||
|
||||
def test_memory_chunks_route_empty_match_is_200_empty_list():
|
||||
# The 0-chunks state is a VISIBLE answer (not a 404): count 0, total shows the store.
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
app = build_memory_provider_app(store, heimdall_key=b"k")
|
||||
r = TestClient(app).get("/memory/chunks", params={"end_user_id": "vuong"})
|
||||
assert r.status_code == 200
|
||||
body = r.json()
|
||||
assert body == {"chunks": [], "count": 0, "total": 0}
|
||||
|
||||
|
||||
def test_memory_chunks_route_missing_end_user_id_returns_400():
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
app = build_memory_provider_app(store, heimdall_key=b"k")
|
||||
r = TestClient(app).get("/memory/chunks") # no end_user_id
|
||||
assert r.status_code == 400
|
||||
assert r.json()["error_code"] == "missing_end_user_id"
|
||||
|
||||
|
||||
def test_build_memory_app_keeps_bifrost_routes_top_level():
|
||||
# POST-002 parity with affect D2: add_memory_read_route uses add_route (not Mount),
|
||||
# so /bifrost/* stay top-level and the op-feed path check still matches them.
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
app = build_memory_provider_app(store, heimdall_key=b"k")
|
||||
paths = {getattr(r, "path", None) for r in app.routes}
|
||||
assert "/bifrost/handshake" in paths
|
||||
assert "/bifrost/memory-call" in paths
|
||||
assert "/memory/chunks" in paths
|
||||
|
||||
|
||||
def test_add_memory_read_route_is_shared_helper_on_bare_app():
|
||||
# The helper mounts the route on any app (used by both build_memory_provider_app and
|
||||
# the combined provider) — mirror of add_affect_read_route's shared-helper shape.
|
||||
from starlette.applications import Starlette
|
||||
store = open_memory_store(":memory:", embedding_dim=EMBEDDING_DIM)
|
||||
app = Starlette()
|
||||
add_memory_read_route(app, store)
|
||||
r = TestClient(app).get("/memory/chunks", params={"end_user_id": "u"})
|
||||
assert r.status_code == 200
|
||||
assert r.json()["total"] == 0
|
||||
|
||||
+187
-2
@@ -8,6 +8,7 @@ from ratatoskr.sessions import (
|
||||
AgentInfo,
|
||||
AgentNotAvailable,
|
||||
AgentNotFound,
|
||||
AuthoredHistoryUnavailable,
|
||||
AuthScopeDenied,
|
||||
BifrostBinding,
|
||||
BifrostConsumerKeyMissing,
|
||||
@@ -25,11 +26,13 @@ from ratatoskr.sessions import (
|
||||
get_me,
|
||||
get_persona_state,
|
||||
get_session_bifrost,
|
||||
get_session_messages,
|
||||
get_session_tools,
|
||||
list_agents,
|
||||
list_character_models,
|
||||
list_sessions,
|
||||
set_persona_state,
|
||||
write_authored_history,
|
||||
)
|
||||
|
||||
|
||||
@@ -1179,9 +1182,13 @@ class TestSetPersonaState:
|
||||
return_value=httpx.Response(204)
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
result = await set_persona_state(client, "s1", {"pad": [0.4, 0.1, -0.2]})
|
||||
result = await set_persona_state(
|
||||
client, "s1", {"pad": {"pleasure": 0.4, "arousal": 0.1, "dominance": -0.2}}
|
||||
)
|
||||
assert result is None
|
||||
assert _json.loads(route.calls[0].request.content) == {"pad": [0.4, 0.1, -0.2]}
|
||||
assert _json.loads(route.calls[0].request.content) == {
|
||||
"pad": {"pleasure": 0.4, "arousal": 0.1, "dominance": -0.2}
|
||||
}
|
||||
|
||||
@respx.mock
|
||||
async def test_non_204_raises(self) -> None:
|
||||
@@ -1193,3 +1200,181 @@ class TestSetPersonaState:
|
||||
with pytest.raises(SessionApiFailed) as exc:
|
||||
await set_persona_state(client, "s1", {"pad": [1, 2, 3]})
|
||||
assert exc.value.status == 422
|
||||
|
||||
|
||||
_AUTHORED_ACK = {
|
||||
"author": "assistant",
|
||||
"content_chars": 5,
|
||||
"injected_at": "2026-07-06T12:00:00+00:00",
|
||||
"phase": "seeded",
|
||||
"seq": 0,
|
||||
"session_id": "s1",
|
||||
"turn_id": "t1",
|
||||
}
|
||||
|
||||
|
||||
class TestWriteAuthoredHistory:
|
||||
"""write_authored_history — #347 POST /sessions/{id}/history (contract #2 amendment)."""
|
||||
|
||||
@respx.mock
|
||||
async def test_happy_fresh_201(self) -> None:
|
||||
"""happy_fresh_201 [happy,tracer]: 201 → ack verbatim; minimal body."""
|
||||
import json as _json
|
||||
|
||||
route = respx.post("https://w.example/sessions/s1/history").mock(
|
||||
return_value=httpx.Response(201, json=_AUTHORED_ACK)
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
result = await write_authored_history(
|
||||
client, "s1", content="hello", idempotency_key="k1"
|
||||
)
|
||||
assert result == _AUTHORED_ACK
|
||||
assert _json.loads(route.calls[0].request.content) == {
|
||||
"author": "assistant",
|
||||
"content": "hello",
|
||||
"idempotency_key": "k1",
|
||||
}
|
||||
|
||||
@respx.mock
|
||||
async def test_happy_replay_200(self) -> None:
|
||||
"""happy_replay_200 [happy]: 200 replay (byte-identical body) → dict verbatim."""
|
||||
respx.post("https://w.example/sessions/s1/history").mock(
|
||||
return_value=httpx.Response(200, json=_AUTHORED_ACK)
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
result = await write_authored_history(
|
||||
client, "s1", content="hello", idempotency_key="k1"
|
||||
)
|
||||
assert result == _AUTHORED_ACK
|
||||
|
||||
@respx.mock
|
||||
async def test_body_includes_effects(self) -> None:
|
||||
"""body_includes_effects [trace]: effects + claimed_original_at appear iff non-None."""
|
||||
import json as _json
|
||||
|
||||
route = respx.post("https://w.example/sessions/s1/history").mock(
|
||||
return_value=httpx.Response(201, json=_AUTHORED_ACK)
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
await write_authored_history(
|
||||
client,
|
||||
"s1",
|
||||
content="hi",
|
||||
idempotency_key="k1",
|
||||
effects="none",
|
||||
claimed_original_at="2020-01-01T00:00:00Z",
|
||||
)
|
||||
assert _json.loads(route.calls[0].request.content) == {
|
||||
"author": "assistant",
|
||||
"content": "hi",
|
||||
"idempotency_key": "k1",
|
||||
"effects": "none",
|
||||
"claimed_original_at": "2020-01-01T00:00:00Z",
|
||||
}
|
||||
|
||||
@respx.mock
|
||||
async def test_hide_existence_404(self) -> None:
|
||||
"""hide_existence_404 [error]: 404 → AuthoredHistoryUnavailable (NOT SessionApiFailed)."""
|
||||
respx.post("https://w.example/sessions/s1/history").mock(
|
||||
return_value=httpx.Response(404, json={"error_code": "session_not_found"})
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
with pytest.raises(AuthoredHistoryUnavailable) as exc:
|
||||
await write_authored_history(client, "s1", content="hi", idempotency_key="k1")
|
||||
assert exc.value.session_id == "s1"
|
||||
|
||||
@respx.mock
|
||||
async def test_generation_active_409(self) -> None:
|
||||
"""generation_active_409 [error]: 409 → SessionApiFailed(409)."""
|
||||
respx.post("https://w.example/sessions/s1/history").mock(
|
||||
return_value=httpx.Response(409, json={"error_code": "generation_active"})
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
with pytest.raises(SessionApiFailed) as exc:
|
||||
await write_authored_history(client, "s1", content="hi", idempotency_key="k1")
|
||||
assert exc.value.status == 409
|
||||
|
||||
@respx.mock
|
||||
async def test_content_too_long_422(self) -> None:
|
||||
"""content_too_long_422 [error]: 422 → SessionApiFailed(422)."""
|
||||
respx.post("https://w.example/sessions/s1/history").mock(
|
||||
return_value=httpx.Response(422, json={"error_code": "content_too_long"})
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
with pytest.raises(SessionApiFailed) as exc:
|
||||
await write_authored_history(client, "s1", content="x", idempotency_key="k1")
|
||||
assert exc.value.status == 422
|
||||
|
||||
@respx.mock
|
||||
async def test_empty_content(self) -> None:
|
||||
"""empty_content [adversarial]: content="" → AssertionError; no HTTP issued."""
|
||||
route = respx.post("https://w.example/sessions/s1/history").mock(
|
||||
return_value=httpx.Response(201, json=_AUTHORED_ACK)
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
with pytest.raises(AssertionError):
|
||||
await write_authored_history(client, "s1", content="", idempotency_key="k1")
|
||||
assert not route.called
|
||||
|
||||
@respx.mock
|
||||
async def test_empty_idempotency_key(self) -> None:
|
||||
"""empty_idempotency_key [adversarial]: key="" → AssertionError; no HTTP issued."""
|
||||
route = respx.post("https://w.example/sessions/s1/history").mock(
|
||||
return_value=httpx.Response(201, json=_AUTHORED_ACK)
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
with pytest.raises(AssertionError):
|
||||
await write_authored_history(client, "s1", content="hi", idempotency_key="")
|
||||
assert not route.called
|
||||
|
||||
@respx.mock
|
||||
async def test_empty_session_id(self) -> None:
|
||||
"""empty_session_id [adversarial]: session_id="" → AssertionError; no HTTP issued."""
|
||||
route = respx.post("https://w.example/sessions/s1/history").mock(
|
||||
return_value=httpx.Response(201, json=_AUTHORED_ACK)
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
with pytest.raises(AssertionError):
|
||||
await write_authored_history(client, "", content="hi", idempotency_key="k1")
|
||||
assert not route.called
|
||||
|
||||
|
||||
class TestGetSessionMessages:
|
||||
"""#2 contract (amendment 2026-07-06) — get_session_messages (GET /sessions/{id}/messages)."""
|
||||
|
||||
@respx.mock
|
||||
async def test_happy(self) -> None:
|
||||
"""happy [happy,tracer]: 200 {session_id, items, next_cursor} → dict verbatim."""
|
||||
payload = {
|
||||
"session_id": "s1",
|
||||
"items": [{"seq": 0, "role": "assistant", "content": "hello there"}],
|
||||
"next_cursor": None,
|
||||
}
|
||||
respx.get("https://w.example/sessions/s1/messages").mock(
|
||||
return_value=httpx.Response(200, json=payload)
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
result = await get_session_messages(client, "s1")
|
||||
assert result == payload
|
||||
|
||||
@respx.mock
|
||||
async def test_not_found_404(self) -> None:
|
||||
"""not_found_404 [error]: 404 → SessionApiFailed(404)."""
|
||||
respx.get("https://w.example/sessions/s1/messages").mock(
|
||||
return_value=httpx.Response(404, json={"error_code": "session_not_found"})
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
with pytest.raises(SessionApiFailed) as exc:
|
||||
await get_session_messages(client, "s1")
|
||||
assert exc.value.status == 404
|
||||
|
||||
@respx.mock
|
||||
async def test_empty_session_id(self) -> None:
|
||||
"""empty_session_id [adversarial]: "" → AssertionError; no HTTP issued."""
|
||||
route = respx.get("https://w.example/sessions/s1/messages").mock(
|
||||
return_value=httpx.Response(200, json={})
|
||||
)
|
||||
async with httpx.AsyncClient(base_url="https://w.example") as client:
|
||||
with pytest.raises(AssertionError):
|
||||
await get_session_messages(client, "")
|
||||
assert not route.called
|
||||
|
||||
@@ -2922,6 +2922,10 @@ class TestTuiBifrostBind:
|
||||
},
|
||||
)
|
||||
)
|
||||
# sindra is a preset agent → the TUI create path now auto-seeds a #347 first-message.
|
||||
respx.post("https://w.example/sessions/s-bound/history").mock(
|
||||
return_value=httpx.Response(201, json={})
|
||||
)
|
||||
|
||||
async def fake_run_async(self) -> int:
|
||||
return 0
|
||||
|
||||
+299
-1
@@ -175,6 +175,62 @@ class TestCreateSessionEndpoint:
|
||||
resp = TestClient(app).post("/api/sessions", json={})
|
||||
assert resp.status_code == 400
|
||||
|
||||
@respx.mock
|
||||
def test_preset_agent_auto_seeds_first_message(self) -> None:
|
||||
"""#347: a preset agent gets its opening seeded on create; a non-preset agent does not."""
|
||||
respx.post("https://w.example/sessions").mock(return_value=httpx.Response(201, json=_CREATE_OK))
|
||||
hist = respx.post("https://w.example/sessions/s-1/history").mock(
|
||||
return_value=httpx.Response(
|
||||
201,
|
||||
json={
|
||||
"author": "assistant", "seq": 0, "phase": "seeded", "turn_id": "t1",
|
||||
"session_id": "s-1", "content_chars": 1, "injected_at": "t",
|
||||
},
|
||||
)
|
||||
)
|
||||
from ratatoskr.web.server import create_app
|
||||
app = create_app(_mock_client_factory())
|
||||
client = TestClient(app)
|
||||
# preset agent → the endpoint seeds a first-message
|
||||
assert client.post("/api/sessions", json={"agent_id": "ratatoskr:sindra"}).status_code == 201
|
||||
assert hist.call_count == 1
|
||||
# non-preset agent → no seed (count unchanged)
|
||||
assert client.post("/api/sessions", json={"agent_id": "mimir"}).status_code == 201
|
||||
assert hist.call_count == 1
|
||||
|
||||
|
||||
class TestSessionMessagesEndpoint:
|
||||
"""GET /api/sessions/{id}/messages — proxy session history (renders the #347 seed)."""
|
||||
|
||||
@respx.mock
|
||||
def test_happy_returns_history(self) -> None:
|
||||
"""happy [tracer]: proxies GET /sessions/{id}/messages → 200 with the items verbatim."""
|
||||
payload = {
|
||||
"session_id": "s-1",
|
||||
"items": [{"seq": 0, "role": "assistant", "content": "Hey there."}],
|
||||
"next_cursor": None,
|
||||
}
|
||||
respx.get("https://w.example/sessions/s-1/messages").mock(
|
||||
return_value=httpx.Response(200, json=payload)
|
||||
)
|
||||
from ratatoskr.web.server import create_app
|
||||
app = create_app(_mock_client_factory())
|
||||
resp = TestClient(app).get("/api/sessions/s-1/messages")
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["items"][0]["content"] == "Hey there."
|
||||
|
||||
@respx.mock
|
||||
def test_non_200_status_preserved(self) -> None:
|
||||
"""error: upstream 404 → status-preserving session_messages_unavailable envelope."""
|
||||
respx.get("https://w.example/sessions/ghost/messages").mock(
|
||||
return_value=httpx.Response(404, json={"error_code": "session_not_found"})
|
||||
)
|
||||
from ratatoskr.web.server import create_app
|
||||
app = create_app(_mock_client_factory())
|
||||
resp = TestClient(app).get("/api/sessions/ghost/messages")
|
||||
assert resp.status_code == 404
|
||||
assert resp.json()["error_code"] == "session_messages_unavailable"
|
||||
|
||||
|
||||
_SNAPSHOT = {
|
||||
"agent_id": "mimir",
|
||||
@@ -625,6 +681,11 @@ class TestCreateAppShape:
|
||||
"/", "/version", "/api/agents", "/api/sessions",
|
||||
"/api/agents/{agent_id}/persona_state",
|
||||
"/api/affect/{agent_id}",
|
||||
"/api/memory/chunks",
|
||||
# v0.19.2 debug-surface parity (create_app POST-002)
|
||||
"/api/sessions/{session_id}/tools",
|
||||
"/api/sessions/{session_id}/bifrost",
|
||||
"/api/admin/events",
|
||||
"/api/turns/{session_id}", "/api/turns/{session_id}/stream",
|
||||
"/api/turns/{session_id}/cancel",
|
||||
):
|
||||
@@ -633,10 +694,14 @@ class TestCreateAppShape:
|
||||
assert "/static" in paths
|
||||
|
||||
def test_state_attached(self) -> None:
|
||||
"""state_attached [trace]: app.state.turn_registry is empty dict."""
|
||||
"""state_attached [trace]: app.state.turn_registry is empty dict; admin_key stored."""
|
||||
from ratatoskr.web.server import create_app
|
||||
app = create_app(_mock_client_factory())
|
||||
assert app.state.turn_registry == {}
|
||||
# create_app POST-001: admin_key defaults None (admin routes fail-visible)
|
||||
assert app.state.admin_key is None
|
||||
app2 = create_app(_mock_client_factory(), admin_key="adm-key")
|
||||
assert app2.state.admin_key == "adm-key"
|
||||
|
||||
def test_factory_stored(self) -> None:
|
||||
"""factory_stored [trace]: app.state.client_factory is the same callable."""
|
||||
@@ -813,6 +878,10 @@ class TestWebBifrostBind:
|
||||
route = respx.post("https://w.example/sessions").mock(
|
||||
return_value=httpx.Response(201, json=_CREATE_OK)
|
||||
)
|
||||
# sindra is a preset agent → the endpoint now auto-seeds a #347 first-message.
|
||||
respx.post("https://w.example/sessions/s-1/history").mock(
|
||||
return_value=httpx.Response(201, json={})
|
||||
)
|
||||
app = create_app(
|
||||
_mock_client_factory(),
|
||||
bifrost_consumer_key="server-ck",
|
||||
@@ -848,6 +917,10 @@ class TestWebBifrostBind:
|
||||
route = respx.post("https://w.example/sessions").mock(
|
||||
return_value=httpx.Response(201, json=_CREATE_OK)
|
||||
)
|
||||
# sindra is a preset agent → the endpoint now auto-seeds a #347 first-message.
|
||||
respx.post("https://w.example/sessions/s-1/history").mock(
|
||||
return_value=httpx.Response(201, json={})
|
||||
)
|
||||
app = create_app(
|
||||
_mock_client_factory(),
|
||||
bifrost_consumer_key="server-ck",
|
||||
@@ -1062,3 +1135,228 @@ class TestAffectStateEndpoint:
|
||||
resp = TestClient(app).get("/api/affect/ratatoskr:sindra")
|
||||
assert resp.status_code == 400
|
||||
assert resp.json()["error_code"] == "missing_end_user_id"
|
||||
|
||||
|
||||
class TestSessionToolsEndpoint:
|
||||
"""session_tools_endpoint — proxy owner-scoped GET /sessions/{id}/tools (#183)."""
|
||||
|
||||
@respx.mock
|
||||
def test_happy_returns_inventory(self) -> None:
|
||||
"""happy [tracer]: 200 inventory → 200 verbatim."""
|
||||
respx.get("https://w.example/sessions/s-1/tools").mock(
|
||||
return_value=httpx.Response(200, json={
|
||||
"agent_id": "ratatoskr:sindra",
|
||||
"builtin_tools": ["echo"],
|
||||
"bifrost_tools": [{"name": "memory.search"}],
|
||||
})
|
||||
)
|
||||
from ratatoskr.web.server import create_app
|
||||
resp = TestClient(create_app(_mock_client_factory())).get("/api/sessions/s-1/tools")
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["agent_id"] == "ratatoskr:sindra"
|
||||
|
||||
@respx.mock
|
||||
def test_upstream_404_status_preserving_envelope(self) -> None:
|
||||
"""error: upstream 404 → 404 session_tools_unavailable envelope."""
|
||||
respx.get("https://w.example/sessions/s-1/tools").mock(
|
||||
return_value=httpx.Response(404, content=b"nope")
|
||||
)
|
||||
from ratatoskr.web.server import create_app
|
||||
resp = TestClient(create_app(_mock_client_factory())).get("/api/sessions/s-1/tools")
|
||||
assert resp.status_code == 404
|
||||
assert resp.json()["error_code"] == "session_tools_unavailable"
|
||||
|
||||
|
||||
class TestSessionBifrostEndpoint:
|
||||
"""session_bifrost_endpoint — proxy admin-scoped GET /admin/sessions/{id}/bifrost (#176)."""
|
||||
|
||||
@respx.mock
|
||||
def test_happy_overrides_with_admin_bearer(self) -> None:
|
||||
"""happy [tracer]: 200 state → 200; request carries the ADMIN bearer, not consumer."""
|
||||
route = respx.get("https://w.example/admin/sessions/s-1/bifrost").mock(
|
||||
return_value=httpx.Response(200, json={
|
||||
"endpoint_url": "http://x:8392", "connected": True,
|
||||
"capabilities_granted": ["memory", "affect"], "tools": [],
|
||||
})
|
||||
)
|
||||
from ratatoskr.web.server import create_app
|
||||
app = create_app(_mock_client_factory(), admin_key="adm-key")
|
||||
resp = TestClient(app).get("/api/sessions/s-1/bifrost")
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["connected"] is True
|
||||
assert route.calls.last.request.headers["Authorization"] == "Bearer adm-key"
|
||||
|
||||
def test_no_admin_key_fails_visible_400(self) -> None:
|
||||
"""error: no admin key configured → 400 admin_key_not_configured, no upstream call."""
|
||||
from ratatoskr.web.server import create_app
|
||||
app = create_app(_mock_client_factory()) # no admin_key
|
||||
resp = TestClient(app).get("/api/sessions/s-1/bifrost")
|
||||
assert resp.status_code == 400
|
||||
assert resp.json()["error_code"] == "admin_key_not_configured"
|
||||
|
||||
@respx.mock
|
||||
def test_upstream_404_status_preserving_envelope(self) -> None:
|
||||
"""error: upstream 404 (not bound) → 404 bifrost_state_unavailable envelope."""
|
||||
respx.get("https://w.example/admin/sessions/s-1/bifrost").mock(
|
||||
return_value=httpx.Response(404, content=b"nope")
|
||||
)
|
||||
from ratatoskr.web.server import create_app
|
||||
app = create_app(_mock_client_factory(), admin_key="adm-key")
|
||||
resp = TestClient(app).get("/api/sessions/s-1/bifrost")
|
||||
assert resp.status_code == 404
|
||||
assert resp.json()["error_code"] == "bifrost_state_unavailable"
|
||||
|
||||
|
||||
class TestAdminEventsEndpoint:
|
||||
"""admin_events_endpoint — SSE proxy of GET /admin/events, session-filtered (#11)."""
|
||||
|
||||
def test_filter_semantics(self) -> None:
|
||||
"""unit: heartbeats drop, system.* pass, else match on session_id."""
|
||||
from ratatoskr.sse_client import AdminEvent
|
||||
from ratatoskr.web.server import _admin_event_matches_web
|
||||
|
||||
def mk(t: str, sid: "str | None" = None) -> AdminEvent:
|
||||
return AdminEvent(id=1, type=t, timestamp=None,
|
||||
data={"session_id": sid} if sid else {})
|
||||
|
||||
assert _admin_event_matches_web(mk("system.heartbeat"), "s-1") is False
|
||||
assert _admin_event_matches_web(mk("system.degraded"), "s-1") is True
|
||||
assert _admin_event_matches_web(mk("session.created", "s-1"), "s-1") is True
|
||||
assert _admin_event_matches_web(mk("session.created", "other"), "s-1") is False
|
||||
assert _admin_event_matches_web(mk("session.created", "s-1"), None) is False
|
||||
|
||||
def test_no_admin_key_fails_visible_400(self) -> None:
|
||||
"""error: no admin key → 400 admin_key_not_configured (no stream opened)."""
|
||||
from ratatoskr.web.server import create_app
|
||||
app = create_app(_mock_client_factory())
|
||||
resp = TestClient(app).get("/api/admin/events?session_id=s-1")
|
||||
assert resp.status_code == 400
|
||||
assert resp.json()["error_code"] == "admin_key_not_configured"
|
||||
|
||||
@respx.mock
|
||||
def test_streams_filtered_events_fixed_name(self) -> None:
|
||||
"""happy: SSE → only session-matching + system.* forwarded, as `admin_event`."""
|
||||
stream = (
|
||||
b'event: session.created\n'
|
||||
b'data: {"type":"session.created","data":{"session_id":"s-1"}}\n\n'
|
||||
b'event: system.heartbeat\n'
|
||||
b'data: {"type":"system.heartbeat","data":{}}\n\n'
|
||||
b'event: turn.started\n'
|
||||
b'data: {"type":"turn.started","data":{"session_id":"other"}}\n\n'
|
||||
b'event: system.degraded\n'
|
||||
b'data: {"type":"system.degraded","data":{}}\n\n'
|
||||
)
|
||||
respx.get("https://w.example/admin/events").mock(return_value=_sse_resp(stream))
|
||||
from ratatoskr.web.server import create_app
|
||||
app = create_app(_mock_client_factory(), admin_key="adm-key")
|
||||
body = TestClient(app).get("/api/admin/events?session_id=s-1").text
|
||||
assert "event: admin_event" in body # fixed browser-facing name
|
||||
assert '"type": "session.created"' in body # matches active session → forwarded
|
||||
assert "system.degraded" in body # system.* → forwarded
|
||||
assert "system.heartbeat" not in body # heartbeat → dropped
|
||||
assert "turn.started" not in body # other session → dropped
|
||||
|
||||
@respx.mock
|
||||
def test_stream_error_on_connect_failure(self) -> None:
|
||||
"""error: upstream admin SSE non-200 -> ONE stream_error frame, stream ends (POST-003)."""
|
||||
respx.get("https://w.example/admin/events").mock(
|
||||
return_value=httpx.Response(500, content=b"boom")
|
||||
)
|
||||
from ratatoskr.web.server import create_app
|
||||
app = create_app(_mock_client_factory(), admin_key="adm-key")
|
||||
body = TestClient(app).get("/api/admin/events?session_id=s-1").text
|
||||
assert "event: stream_error" in body
|
||||
assert "SseConnectFailed" in body
|
||||
assert body.count("event: stream_error") == 1 # exactly one, then ends
|
||||
|
||||
|
||||
class TestMemoryChunksEndpoint:
|
||||
"""memory_chunks_endpoint FN — memory viewer: web proxy to the provider debug read.
|
||||
Mirrors TestAffectStateEndpoint (the #18-D2 affect proxy shape)."""
|
||||
|
||||
@respx.mock
|
||||
def test_happy_proxies_and_supplies_server_end_user_id(self) -> None:
|
||||
"""tracer: GET /api/memory/chunks → proxies to the configured provider read URL,
|
||||
supplying end_user_id SERVER-SIDE and forwarding the browser-named agent_id."""
|
||||
from ratatoskr.web.server import create_app
|
||||
|
||||
payload = {
|
||||
"chunks": [
|
||||
{"chunk_id": "c1", "content": "the user's cat is Mochi",
|
||||
"scope": {"end_user": "vuong", "agent_self": "ratatoskr:sindra"},
|
||||
"origin": "worldtree", "revision": 1}
|
||||
],
|
||||
"count": 1,
|
||||
"total": 1,
|
||||
}
|
||||
route = respx.get(url__regex=r"http://prov:8391/memory/chunks.*").mock(
|
||||
return_value=httpx.Response(200, json=payload)
|
||||
)
|
||||
app = create_app(
|
||||
_mock_client_factory(),
|
||||
end_user_id="vuong",
|
||||
memory_read_url="http://prov:8391",
|
||||
)
|
||||
resp = TestClient(app).get("/api/memory/chunks?agent_id=ratatoskr:sindra")
|
||||
assert resp.status_code == 200
|
||||
assert resp.json() == payload
|
||||
assert route.calls.last.request.url.params["end_user_id"] == "vuong"
|
||||
assert route.calls.last.request.url.params["agent_id"] == "ratatoskr:sindra"
|
||||
|
||||
@respx.mock
|
||||
def test_browser_supplied_end_user_id_is_ignored(self) -> None:
|
||||
"""The server's configured partition is used; a browser end_user_id is ignored."""
|
||||
from ratatoskr.web.server import create_app
|
||||
|
||||
route = respx.get(url__regex=r"http://prov:8391/memory/chunks.*").mock(
|
||||
return_value=httpx.Response(200, json={"chunks": [], "count": 0, "total": 0})
|
||||
)
|
||||
app = create_app(
|
||||
_mock_client_factory(), end_user_id="vuong", memory_read_url="http://prov:8391"
|
||||
)
|
||||
TestClient(app).get("/api/memory/chunks?end_user_id=attacker&agent_id=a")
|
||||
assert route.calls.last.request.url.params["end_user_id"] == "vuong"
|
||||
|
||||
def test_unconfigured_returns_400(self) -> None:
|
||||
"""PRE-001: no memory_read_url → 400 memory_not_configured (no silent attempt)."""
|
||||
from ratatoskr.web.server import create_app
|
||||
|
||||
app = create_app(_mock_client_factory(), end_user_id="vuong") # no memory_read_url
|
||||
resp = TestClient(app).get("/api/memory/chunks?agent_id=a")
|
||||
assert resp.status_code == 400
|
||||
assert resp.json()["error_code"] == "memory_not_configured"
|
||||
|
||||
def test_no_end_user_configured_returns_400(self) -> None:
|
||||
from ratatoskr.web.server import create_app
|
||||
|
||||
app = create_app(_mock_client_factory(), memory_read_url="http://prov:8391")
|
||||
resp = TestClient(app).get("/api/memory/chunks?agent_id=a")
|
||||
assert resp.status_code == 400
|
||||
assert resp.json()["error_code"] == "memory_not_configured"
|
||||
|
||||
@respx.mock
|
||||
def test_provider_unreachable_returns_502(self) -> None:
|
||||
from ratatoskr.web.server import create_app
|
||||
|
||||
respx.get(url__regex=r"http://prov:8391/memory/chunks.*").mock(
|
||||
side_effect=httpx.ConnectError("refused")
|
||||
)
|
||||
app = create_app(
|
||||
_mock_client_factory(), end_user_id="vuong", memory_read_url="http://prov:8391"
|
||||
)
|
||||
resp = TestClient(app).get("/api/memory/chunks?agent_id=a")
|
||||
assert resp.status_code == 502
|
||||
assert resp.json()["error_code"] == "memory_provider_unreachable"
|
||||
|
||||
@respx.mock
|
||||
def test_provider_400_passes_through(self) -> None:
|
||||
from ratatoskr.web.server import create_app
|
||||
|
||||
respx.get(url__regex=r"http://prov:8391/memory/chunks.*").mock(
|
||||
return_value=httpx.Response(400, json={"error_code": "missing_end_user_id"})
|
||||
)
|
||||
app = create_app(
|
||||
_mock_client_factory(), end_user_id="vuong", memory_read_url="http://prov:8391"
|
||||
)
|
||||
resp = TestClient(app).get("/api/memory/chunks?agent_id=a")
|
||||
assert resp.status_code == 400
|
||||
|
||||
@@ -190,14 +190,14 @@ wheels = [
|
||||
|
||||
[[package]]
|
||||
name = "bifrost"
|
||||
version = "1.0.0"
|
||||
version = "1.1.4"
|
||||
source = { registry = "https://gitea.phasefinal.com/api/packages/vh/pypi/simple/" }
|
||||
dependencies = [
|
||||
{ name = "jsonschema" },
|
||||
]
|
||||
sdist = { url = "https://gitea.phasefinal.com/api/packages/vh/pypi/files/bifrost/1.0.0/bifrost-1.0.0.tar.gz", hash = "sha256:93130d68dfd9868580a4514277996ba176837972b9e42129eda8bb03ad3b18b9" }
|
||||
sdist = { url = "https://gitea.phasefinal.com/api/packages/vh/pypi/files/bifrost/1.1.4/bifrost-1.1.4.tar.gz", hash = "sha256:498d156035a93bf37a6fc1e9c09b468aac61e869fd2a5353e2695dc823f57e9a" }
|
||||
wheels = [
|
||||
{ url = "https://gitea.phasefinal.com/api/packages/vh/pypi/files/bifrost/1.0.0/bifrost-1.0.0-py3-none-any.whl", hash = "sha256:1a53baa2b0596b7c418e2d82e3eeee0f13054604d78b592609ee1aff90dccac2" },
|
||||
{ url = "https://gitea.phasefinal.com/api/packages/vh/pypi/files/bifrost/1.1.4/bifrost-1.1.4-py3-none-any.whl", hash = "sha256:d67278528f12729eef0c19d875d36a2f1da6fa97737396ba130d525fee8d0b14" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -1052,7 +1052,7 @@ wheels = [
|
||||
|
||||
[[package]]
|
||||
name = "ratatoskr"
|
||||
version = "0.19.1"
|
||||
version = "0.20.15"
|
||||
source = { editable = "." }
|
||||
dependencies = [
|
||||
{ name = "httpx" },
|
||||
@@ -1086,7 +1086,7 @@ web = [
|
||||
|
||||
[package.metadata]
|
||||
requires-dist = [
|
||||
{ name = "bifrost", marker = "extra == 'provider'", specifier = "==1.0.0", index = "https://gitea.phasefinal.com/api/packages/vh/pypi/simple/" },
|
||||
{ name = "bifrost", marker = "extra == 'provider'", specifier = "==1.1.4", index = "https://gitea.phasefinal.com/api/packages/vh/pypi/simple/" },
|
||||
{ name = "httpx", specifier = ">=0.27" },
|
||||
{ name = "httpx-sse", specifier = ">=0.4" },
|
||||
{ name = "jsonschema", marker = "extra == 'provider'", specifier = ">=4" },
|
||||
|
||||
Reference in New Issue
Block a user