diff --git a/docs/contracts/issues/13.contract.md b/docs/contracts/issues/13.contract.md index a357fbe..50c2d4a 100644 --- a/docs/contracts/issues/13.contract.md +++ b/docs/contracts/issues/13.contract.md @@ -158,8 +158,12 @@ New `Static(id="pane-name")` widget alongside the existing `identity` + `hint` w - **INV-016**: Input retains keyboard focus across `Ctrl+1` / `Ctrl+2` tab switches. - **INV-017** *(amended v0.5.0)*: `thinking-current` Static docks to the top of the **right column** (above `TabbedContent`), not the left column. Live thinking visibility persists across tab switches. v0.5.0 moves it from left → right so the left column is genuinely content-only. - **INV-018**: CLI mode (`ratatoskr.cli._amain`) is unaffected. CLI keeps inline `· tool_start: …` / `· tool_result: …` rendering on stderr per issue #12 INV-005. -- **INV-019** *(new v0.5.0)*: Two TabPanes in the right column: `Tools` (id `tools-tab`, contains `#tools-log`) + `Debug` (id `debug-tab`, contains `#debug-log`). Ctrl+1 activates Tools; Ctrl+2 activates Debug. `pane-name` Static reflects the active tab name dynamically. -- **INV-020** *(new v0.5.0)*: Render-exception fallback (INV-009) preserves routing per event class: `ToolStart` / `ToolResult` fallback writes to `tools_log`; `WorkerPhase` / `Thinking` / `TextBoundary` fallback writes to `debug_log`; everything else falls back to `log`. +- **INV-019** *(amended v0.6.0)*: Three TabPanes in the right column: `Tools` (id `tools-tab`, contains `#tools-log`) + `Debug` (id `debug-tab`, contains `#debug-log`) + `Thinking` (id `thinking-tab`, contains `#thinking-log`). Ctrl+1/Ctrl+2/Ctrl+3 activate respective tabs. `pane-name` Static reflects active tab name dynamically. +- **INV-020** *(amended v0.6.0)*: Render-exception fallback (INV-009) preserves routing per event class: `ToolStart` / `ToolResult` → `tools_log`; `Thinking` → `thinking_log`; `WorkerPhase` / `TextBoundary` → `debug_log`; everything else → `log`. +- **INV-021** *(new v0.6.0)*: `Text` events do NOT route to `log` per-delta. They accumulate into `TuiPresenterState.text_buffer` and update a single `current_text` Static (docked above the prompt). On terminal event (`Done`/`Error`/`Cancelled`), `current_text` is cleared and (raw mode) accumulated text or (non-raw) post-Done `Markdown(response)` is written to `log`. The pre-v0.6.0 per-token RichLog spam is retired. +- **INV-022** *(new v0.6.0)*: Closed thinking runs route to `thinking_log`, NOT `debug_log`. Each closed run writes three entries: `Rule(title=f"turn N · thinking #K start")`, `Markdown(content)`, `Rule(title=f"turn N · thinking #K end")` — the model's chain-of-thought is presented as rendered Markdown (model reasoning often has lists / code / structure) wrapped in operator-visible start/end markers. `thinking_run_index` increments per-run within a turn. +- **INV-023** *(new v0.6.0)*: Turn-ID header `Rule(title=f"turn N")` is written to all four log panes (`log`, `tools_log`, `debug_log`, `thinking_log`) by `_stream_turn_worker` on the first event of each turn — enables cross-pane visual correlation during multi-turn debugging. +- **INV-024** *(new v0.6.0)*: `thinking-current` Static remains in the right column (above TabbedContent) per INV-017 — live per-delta thinking visibility persists across tab switches. Prefix `thinking… ` self-identifies the widget contents. ## TESTS (additions / changes to test_tui.py) diff --git a/persistent-memory.md b/persistent-memory.md index ebbb705..21dbeeb 100644 --- a/persistent-memory.md +++ b/persistent-memory.md @@ -32,9 +32,9 @@ separate dev team rather than an in-tree Worldtree tool. ## Current state / in-flight -_As of 2026-05-24 (post-v0.5.1 UI polish pass):_ +_As of 2026-05-24 (post-v0.6.0 streaming + Thinking pane + turn headers):_ -**Status: v0.5.1 shipped.** Nine core issues complete (`sse_client` +**Status: v0.6.0 shipped.** Nine core issues complete (`sse_client` #1, `sessions` #2, `cli` #3, `tui` #4, `--end-user-id` #5, TUI startup error visibility #6, presenter contract semantics amendment #12, startup agent picker #8, §5 layout reshape + Tools pane #13) @@ -51,7 +51,8 @@ Static in the footer (static "Tools" v1; dynamic when more tabs land). CLI mode (--send) unaffected by design — INV-018. Last commits on `main`: -- v0.5.1 style(tui): polish pass — colored terminal labels, placeholders, padding +- v0.6.0 refactor(tui): streaming Static + turn headers + Thinking pane + agent picker multi-line +- `7106af5` style(tui): UI polish pass — terminal label colors, placeholders (v0.5.1) - `ffd22fb` refactor(tui): content-only main pane + Debug tab + chrome dark (v0.5.0) - `2756f5f` style(tui): apply Australis theme to TUI chrome + widgets (v0.4.1) - `24e4371` feat(tui): issue #13 — §5 layout reshape + Tools pane (v0.4.0) diff --git a/pyproject.toml b/pyproject.toml index f946671..7fc8005 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "ratatoskr" -version = "0.5.1" +version = "0.6.0" description = "Worldtree Conversation API debug TUI — multi-pane observability dashboard" readme = "README.md" requires-python = ">=3.12" diff --git a/src/ratatoskr/tui.py b/src/ratatoskr/tui.py index fb0c9a3..e3fbfd8 100644 --- a/src/ratatoskr/tui.py +++ b/src/ratatoskr/tui.py @@ -22,7 +22,6 @@ from textual.widgets import ( Footer, Header, Input, - Label, ListItem, ListView, RichLog, @@ -179,6 +178,11 @@ class TuiPresenterState: thinking_buffer: list[str] = field(default_factory=list) thinking_open: bool = False + # v0.6.0: per-turn streaming text buffer. Text deltas accumulate here + # and update `current_text` Static in place — no per-token RichLog spam. + text_buffer: list[str] = field(default_factory=list) + # v0.6.0: thinking-run counter for turn-scoped start/end markers. + thinking_run_index: int = 0 def render( self, @@ -186,20 +190,26 @@ class TuiPresenterState: *, log: RichLog, thinking_widget: Static, + current_text: Static, tools_log: RichLog, debug_log: RichLog, + thinking_log: RichLog, raw: bool, ) -> None: """Render one Worldtree SSE event with the TUI hierarchy + coalescing. - v0.5.0 routing: main `log` (transcript) is CONTENT-ONLY — Text, - terminal labels ([done] / [error] / [cancelled]), and the - post-Done Markdown render. All telemetry (Thinking closed runs, - WorkerPhase, TextBoundary) routes to `debug_log` (Debug tab); all - tool activity (ToolStart, ToolResult) routes to `tools_log` (Tools - tab). Live thinking deltas continue to update `thinking_widget`. + v0.6.0 routing: + - `log` (transcript) = content only: user-prompt echo (written + outside the presenter), terminal labels, post-Done Markdown body. + - `current_text` (Static below transcript) = live-streaming Text + deltas accumulated into one growing line; cleared on terminal. + - `tools_log` = ToolStart + ToolResult. + - `debug_log` = WorkerPhase + TextBoundary. + - `thinking_log` = closed thinking runs (Markdown + start/end + Rule markers); `thinking_widget` continues to receive live + per-delta updates. - Exceptions are caught at the presenter boundary (INV-009 fallback). + Exceptions caught at the presenter boundary (INV-009 fallback). """ assert isinstance( event, @@ -236,25 +246,48 @@ class TuiPresenterState: thinking_widget.update(f"thinking… {tail}") return # Non-thinking event: close any open thinking run. - # v0.5.0: closed thinking runs land in debug_log (Debug pane), not - # transcript — keeps the main pane content-only. + # v0.6.0: closed thinking runs route to thinking_log (Thinking + # pane) wrapped in `── turn N · thinking start/end ──` Rule + # markers, with the content itself rendered as Markdown (model + # reasoning often has lists, code, structure). if self.thinking_open: full_thinking = "".join(self.thinking_buffer) - debug_log.write(_dim(f"· thinking: {full_thinking}")) + turn_id = event.sse_id.turn_id if hasattr(event, "sse_id") else ( + event.turn_id if hasattr(event, "turn_id") else "?" + ) + self.thinking_run_index += 1 + from rich.markdown import Markdown + from rich.rule import Rule + thinking_log.write(Rule( + title=f"turn {turn_id} · thinking #{self.thinking_run_index} start", + style=_AU_DEMOTED, + )) + thinking_log.write(Markdown(full_thinking)) + thinking_log.write(Rule( + title=f"turn {turn_id} · thinking #{self.thinking_run_index} end", + style=_AU_DEMOTED, + )) self.thinking_buffer.clear() self.thinking_open = False thinking_widget.update("") thinking_widget.display = False # Now render the non-thinking event itself. if isinstance(event, Text): - # Streamed text content — no prefix, no demotion. - log.write(event.content) + # v0.6.0: streaming text accumulates into current_text Static + # — one growing live line, NOT per-delta RichLog entries. + self.text_buffer.append(event.content) + current_text.update("".join(self.text_buffer)) return if isinstance(event, (Done, Error, Cancelled)): - # Terminal events: load-bearing label tinted per outcome. - # v0.5.1 polish: Aurora green / Dawn red / Dawn yellow so the - # turn-terminal status is scannable at a glance vs blending - # with default foreground. + # Terminal event: clear the streaming Static first so the + # live-preview band collapses. Then write the colored label + # + (non-raw) Markdown body / (raw) accumulated plain text + # to the transcript. + accumulated = "".join(self.text_buffer) + self.text_buffer.clear() + current_text.update("") + # Terminal labels tinted per outcome (Aurora green / Dawn red + # / Dawn yellow) for at-a-glance scanning. if isinstance(event, Done): log.write(RichText( f"[done] turn_id={event.sse_id.turn_id} model={event.model} " @@ -262,13 +295,15 @@ class TuiPresenterState: f"usage {_format_usage(event.usage, arrow='→')}", style=_AU_SUCCESS, )) - if not raw: + if raw: + # Raw mode: emit the accumulated streamed text verbatim + # so the operator has a record after the Static clears. + if accumulated: + log.write(accumulated) + else: from rich.markdown import Markdown from rich.rule import Rule - # Rule tinted to match the column border so the - # streamed-text / markdown-render boundary reads as - # part of the chrome family, not a content artifact. log.write(Rule(style=_AU_DEMOTED)) log.write(Markdown(event.response)) elif isinstance(event, Error): @@ -318,14 +353,17 @@ class TuiPresenterState: # the original event AND a render_error line with the class name only # (NO exception message — security clause). Volva F1 fix. # - # v0.5.0: routing-under-failure preservation — fallback writes go + # v0.6.0 routing-under-failure preservation — fallback writes go # to the same destination the successful render would have used: # - ToolStart/ToolResult → tools_log - # - WorkerPhase/Thinking/TextBoundary → debug_log - # - everything else (Text/Done/Error/Cancelled) → log + # - Thinking → thinking_log + # - WorkerPhase/TextBoundary → debug_log + # - everything else → log if isinstance(event, (ToolStart, ToolResult)): target = tools_log - elif isinstance(event, (WorkerPhase, Thinking, TextBoundary)): + elif isinstance(event, Thinking): + target = thinking_log + elif isinstance(event, (WorkerPhase, TextBoundary)): target = debug_log else: target = log @@ -344,6 +382,13 @@ class AgentPickerApp(App[str | None]): """ DEFAULT_CSS = """ + Header { + background: $surface; + color: $au-bright-blue; + } + Footer { + background: $surface; + } #picker-prompt { dock: top; height: 1; @@ -355,9 +400,21 @@ class AgentPickerApp(App[str | None]): height: 1fr; background: $background; } + /* v0.6.0: multi-line agent items. Each ListItem is auto-height so the + full description wraps below the agent_id/name line — no truncation. */ + #agent-list > ListItem { + height: auto; + padding: 1 1; + } #agent-list > ListItem.--highlight { - background: $primary; - color: $au-bright-white; + background: $au-dark-30; + } + .agent-id-line { + color: $au-bright-blue; + text-style: bold; + } + .agent-desc { + color: $au-dark-60; } """ @@ -380,9 +437,16 @@ class AgentPickerApp(App[str | None]): def compose(self) -> ComposeResult: yield Header() yield Static("Pick an agent for the new session:", id="picker-prompt") + # v0.6.0: each ListItem has two Static children — the id/name line + # in bold blue + the wrapped description in muted dark-60. No + # description truncation; tall items breathe so the operator can + # actually read what each agent does. yield ListView( *[ - ListItem(Label(f"{a.agent_id} · {a.name} — {a.description}")) + ListItem( + Static(f"{a.agent_id} · {a.name}", classes="agent-id-line"), + Static(a.description, classes="agent-desc"), + ) for a in self.agents ], id="agent-list", @@ -425,6 +489,9 @@ class RatatoskrApp(App[int]): # Australis theme variables ($primary/$accent/$au-dark-60/$au-bright-cyan/ # etc.) carry colors so a future theme swap rebinds centrally. DEFAULT_CSS = """ + /* Kill Textual's default $primary-blue tinting on chrome — Header, + Footer, ContentTabs strip, active-tab Underline all forced to the + Australis cool-dark palette. */ Header { background: $surface; color: $au-bright-blue; @@ -454,25 +521,39 @@ class RatatoskrApp(App[int]): background: $background; padding: 0 1; } - #tools-log, #debug-log { + /* v0.6.0: streaming-text Static carries in-flight assistant tokens. + Replaces per-token RichLog spam — one growing line that updates in + place. Cleared on terminal event; final Markdown body lands in the + transcript. */ + #current-text { + dock: bottom; + height: auto; background: $background; padding: 0 1; } - #side-panes Tabs { + #tools-log, #debug-log, #thinking-log { + background: $background; + padding: 0 1; + } + /* Tab strip + active-tab underline — kill blue, use Australis cyan. */ + #side-panes > ContentTabs { background: $surface; } - /* Active tab: Aurora bright-cyan label so the operator's eye lands - on the currently selected pane name. */ - #side-panes Tab.-active { + #side-panes ContentTab.-active { color: $au-bright-cyan; text-style: bold; } + #side-panes Underline > .underline--bar { + color: $au-bright-cyan; + } #prompt { dock: bottom; border: tall $panel; } + /* v0.6.0: focused border uses Australis bright-cyan instead of $primary + (Aurora blue) — kills the lingering blue tint the user flagged. */ #prompt:focus { - border: tall $primary; + border: tall $au-bright-cyan; } /* Placeholder text in the Input — dimmer than typed content. */ #prompt > .input--placeholder { @@ -505,6 +586,7 @@ class RatatoskrApp(App[int]): # without losing Input focus (INV-016). Binding("ctrl+1", "focus_tools", "Tools tab", priority=False), Binding("ctrl+2", "focus_debug", "Debug tab", priority=False), + Binding("ctrl+3", "focus_thinking", "Thinking tab", priority=False), ] HINT_IDLE = "Ctrl-C twice to exit" @@ -533,17 +615,22 @@ class RatatoskrApp(App[int]): def compose(self) -> ComposeResult: yield Header() - # v0.5.0 layout: left column is content-only (transcript + prompt). - # Right column houses ALL telemetry — thinking-current live preview - # docked above the TabbedContent; tabs cycle Tools / Debug. - # markup=False on RichLog so labeled lines like "[cancel_failed] ..." - # render verbatim; Rich would otherwise interpret bracket spans as - # style markup. The post-Done markdown render uses Markdown() directly - # which is a Rich Renderable and renders correctly without - # widget-level markup=True. + # v0.6.0 layout: left column is content-only (transcript + streaming + # text Static + prompt). Right column hosts thinking-current live + # preview above TabbedContent cycling Tools / Debug / Thinking. + # + # The current-text Static buffers in-flight assistant tokens so + # streaming doesn't spam the RichLog with one line per delta — + # the operator sees a single growing live line, then on Done the + # Static clears and the final Markdown body lands in the transcript. + # + # markup=False on RichLog so labeled lines render verbatim; the + # post-Done Markdown() / Rule() renders are Rich Renderables and + # work without widget-level markup=True. with Horizontal(id="main-row"): with Vertical(id="left-column"): yield RichLog(id="transcript", wrap=True, markup=False, highlight=False) + yield Static("", id="current-text") yield Input(id="prompt", placeholder="Type a message and press Enter") with Vertical(id="right-column"): yield Static("", id="thinking-current") @@ -556,6 +643,10 @@ class RatatoskrApp(App[int]): yield RichLog( id="debug-log", wrap=True, markup=False, highlight=False ) + with TabPane("Thinking", id="thinking-tab"): + yield RichLog( + id="thinking-log", wrap=True, markup=False, highlight=False + ) # INV-002 + INV-003: visible identity + hint widgets (Footer-area). # pane-name widget displays current side-pane name. yield Static("", id="identity") @@ -588,12 +679,36 @@ class RatatoskrApp(App[int]): style=placeholder_style) ) self.query_one("#debug-log", RichLog).write( - RichText("(waiting for telemetry — start a turn)", + RichText("(waiting for worker_phase + text_boundary telemetry)", + style=placeholder_style) + ) + self.query_one("#thinking-log", RichLog).write( + RichText("(no chain-of-thought captured yet — start a turn)", style=placeholder_style) ) self.state = "idle" self._set_hint(self.HINT_IDLE) + def _write_turn_headers(self, turn_id: int) -> None: + """v0.6.0: Write `── turn N ──` Rule headers across every pane so + operators can visually correlate sections during cross-pane + debugging. Called from `_stream_turn_worker` on first event of + each new turn (idempotent per turn via active_turn_id guard). + """ + from rich.rule import Rule + + title = f"turn {turn_id}" + rule = Rule(title=title, style=_AU_DEMOTED) + try: + self.query_one("#transcript", RichLog).write(rule) + self.query_one("#tools-log", RichLog).write(rule) + self.query_one("#debug-log", RichLog).write(rule) + self.query_one("#thinking-log", RichLog).write(rule) + except Exception: + # Defensive: widget tree may be tearing down — never let a + # turn-header write block the SSE consumer. + pass + def _set_hint(self, hint: str) -> None: """Set the hint state attribute AND update the visible Static widget.""" self.hint = hint @@ -633,20 +748,27 @@ class RatatoskrApp(App[int]): assert content log = self.query_one("#transcript", RichLog) thinking_widget = self.query_one("#thinking-current", Static) - # v0.5.0: separate panes for tools vs telemetry; transcript is content only. + current_text = self.query_one("#current-text", Static) tools_log = self.query_one("#tools-log", RichLog) debug_log = self.query_one("#debug-log", RichLog) + thinking_log = self.query_one("#thinking-log", RichLog) presenter = TuiPresenterState() try: async for event in stream_turn(self.client, self.session_id, content): if self.active_turn_id is None: self.active_turn_id = event.sse_id.turn_id + # v0.6.0: turn-ID headers across all panes so the + # operator can visually correlate sections during + # cross-pane debugging. + self._write_turn_headers(self.active_turn_id) presenter.render( event, log=log, thinking_widget=thinking_widget, + current_text=current_text, tools_log=tools_log, debug_log=debug_log, + thinking_log=thinking_log, raw=self.args.raw, ) if isinstance(event, (Done, Error, Cancelled)): @@ -714,6 +836,11 @@ class RatatoskrApp(App[int]): self.query_one("#side-panes", TabbedContent).active = "debug-tab" self.query_one("#pane-name", Static).update("Debug") + def action_focus_thinking(self) -> None: + """v0.6.0: Ctrl+3 activates the Thinking tab. INV-016 preserves Input focus.""" + self.query_one("#side-panes", TabbedContent).active = "thinking-tab" + self.query_one("#pane-name", Static).update("Thinking") + def run_tui(args: ParsedArgs) -> int: """Sync entry point — delegates to the async resolve-then-run flow. diff --git a/tests/test_tui.py b/tests/test_tui.py index 01aadc9..69c395b 100644 --- a/tests/test_tui.py +++ b/tests/test_tui.py @@ -132,7 +132,7 @@ class TestTuiPresenterState: thinking_widget=widget, tools_log=MagicMock(), debug_log=MagicMock(), - raw=False, + current_text=MagicMock(), thinking_log=MagicMock(), raw=False, ) state.render( Thinking(sse_id=SID, content="b"), @@ -140,7 +140,7 @@ class TestTuiPresenterState: thinking_widget=widget, tools_log=MagicMock(), debug_log=MagicMock(), - raw=False, + current_text=MagicMock(), thinking_log=MagicMock(), raw=False, ) state.render( Thinking(sse_id=SID, content="c"), @@ -148,7 +148,7 @@ class TestTuiPresenterState: thinking_widget=widget, tools_log=MagicMock(), debug_log=MagicMock(), - raw=False, + current_text=MagicMock(), thinking_log=MagicMock(), raw=False, ) # Widget updated 3 times — once per delta — with cumulative content assert widget.update.call_count == 3 @@ -160,48 +160,50 @@ class TestTuiPresenterState: # No RichLog write yet — closure hasn't fired assert log.write.call_count == 0 - def test_thinking_closes_one_debuglog_entry(self) -> None: - """thinking_closes_one_debuglog_entry [happy, v0.5.0]: 2x Thinking + WorkerPhase → - debug_log has ONE closed thinking entry + one worker_phase entry; widget cleared+hidden; - transcript (log) untouched. + def test_thinking_closes_to_thinking_log(self) -> None: + """thinking_closes_to_thinking_log [happy, v0.6.0]: 2x Thinking + WorkerPhase → + thinking_log gets Rule(start) + Markdown + Rule(end); debug_log gets worker_phase; + transcript and tools_log untouched. Widget cleared+hidden. """ + from rich.markdown import Markdown + from rich.rule import Rule + from ratatoskr.tui import TuiPresenterState log = MagicMock() debug_log = MagicMock() + thinking_log = MagicMock() widget = MagicMock() state = TuiPresenterState() - state.render( - Thinking(sse_id=SID, content="a"), - log=log, - thinking_widget=widget, - tools_log=MagicMock(), - debug_log=debug_log, - raw=False, - ) - state.render( - Thinking(sse_id=SID, content="b"), - log=log, - thinking_widget=widget, - tools_log=MagicMock(), - debug_log=debug_log, - raw=False, - ) + for content in ("a", "b"): + state.render( + Thinking(sse_id=SID, content=content), + log=log, + thinking_widget=widget, + tools_log=MagicMock(), + debug_log=debug_log, + current_text=MagicMock(), + thinking_log=thinking_log, + raw=False, + ) state.render( WorkerPhase(sse_id=SID, phase="streaming", turn_id=42), log=log, thinking_widget=widget, tools_log=MagicMock(), debug_log=debug_log, + current_text=MagicMock(), + thinking_log=thinking_log, raw=False, ) - # v0.5.0: closure + worker_phase write to debug_log; transcript untouched. - assert debug_log.write.call_count == 2 + # v0.6.0: closure writes Rule(start) + Markdown + Rule(end) to thinking_log. + thinking_writes = [c[0][0] for c in thinking_log.write.call_args_list] + assert any(isinstance(w, Rule) for w in thinking_writes), thinking_writes + assert any(isinstance(w, Markdown) for w in thinking_writes), thinking_writes + # worker_phase still goes to debug_log; transcript still untouched. + assert debug_log.write.called + assert "· worker_phase:" in _text_of(debug_log.write.call_args_list[-1][0][0]) assert not log.write.called - # First write = closed thinking entry containing the full accumulated text - assert "· thinking: ab" in _text_of(debug_log.write.call_args_list[0][0][0]) - # Second write = worker_phase with demotion prefix (also in debug_log) - assert "· worker_phase:" in _text_of(debug_log.write.call_args_list[1][0][0]) # Widget cleared + hidden widget.update.assert_called_with("") assert widget.display is False @@ -221,7 +223,7 @@ class TestTuiPresenterState: thinking_widget=widget, tools_log=MagicMock(), debug_log=MagicMock(), - raw=False, + current_text=MagicMock(), thinking_log=MagicMock(), raw=False, ) last_update = widget.update.call_args_list[-1][0][0] # v0.5.1 polish: widget gets a "thinking… " prefix + ellipsis-truncated tail. @@ -247,7 +249,7 @@ class TestTuiPresenterState: thinking_widget=widget, tools_log=MagicMock(), debug_log=MagicMock(), - raw=False, + current_text=MagicMock(), thinking_log=MagicMock(), raw=False, ) assert widget.display is True # Closure (WorkerPhase) → widget hidden @@ -257,77 +259,64 @@ class TestTuiPresenterState: thinking_widget=widget, tools_log=MagicMock(), debug_log=MagicMock(), - raw=False, + current_text=MagicMock(), thinking_log=MagicMock(), raw=False, ) assert widget.display is False - def test_multiple_thinking_runs_each_get_debuglog_entry(self) -> None: - """multiple_thinking_runs_each_get_debuglog_entry [scenario, v0.5.0]: - Thinking → Text → Thinking → Done → TWO closed thinking entries in debug_log - (transcript receives only the Text + Done content). + def test_multiple_thinking_runs_each_get_thinking_log_section(self) -> None: + """multiple_thinking_runs_each_get_thinking_log_section [scenario, v0.6.0]: + Thinking → Text → Thinking → Done → TWO start/end Rule + Markdown sections + in thinking_log. Text goes to current_text Static (buffered). Transcript + receives [done] label + Markdown body only. """ + from rich.markdown import Markdown + from rich.rule import Rule + from ratatoskr.tui import TuiPresenterState log = MagicMock() - debug_log = MagicMock() + thinking_log = MagicMock() + current_text = MagicMock() widget = MagicMock() state = TuiPresenterState() - state.render( + for evt in ( Thinking(sse_id=SID, content="first"), - log=log, - thinking_widget=widget, - tools_log=MagicMock(), - debug_log=debug_log, - raw=False, - ) - state.render( Text(sse_id=SID, content="hi"), - log=log, - thinking_widget=widget, - tools_log=MagicMock(), - debug_log=debug_log, - raw=False, - ) - state.render( Thinking(sse_id=SID, content="second"), - log=log, - thinking_widget=widget, - tools_log=MagicMock(), - debug_log=debug_log, - raw=False, - ) - # Close the second run with a Done. + ): + state.render( + evt, log=log, thinking_widget=widget, + tools_log=MagicMock(), debug_log=MagicMock(), + current_text=current_text, thinking_log=thinking_log, raw=False, + ) state.render( _make_tui_done(), - log=log, - thinking_widget=widget, - tools_log=MagicMock(), - debug_log=debug_log, - raw=True, + log=log, thinking_widget=widget, + tools_log=MagicMock(), debug_log=MagicMock(), + current_text=current_text, thinking_log=thinking_log, raw=False, ) - # v0.5.0: closed thinking entries land in debug_log, NOT log. - thinking_entries = [_text_of(call[0][0]) for call in debug_log.write.call_args_list] - thinking_entries = [t for t in thinking_entries if t.startswith("· thinking:")] - assert len(thinking_entries) == 2 - assert "first" in thinking_entries[0] - assert "second" in thinking_entries[1] - # transcript receives: "hi" (Text) + "[done] ..." (terminal label) only. + # v0.6.0: thinking_log holds (Rule(start) + Markdown + Rule(end)) x2. + thinking_writes = [c[0][0] for c in thinking_log.write.call_args_list] + rules = [w for w in thinking_writes if isinstance(w, Rule)] + markdowns = [w for w in thinking_writes if isinstance(w, Markdown)] + assert len(rules) == 4, f"expected 4 Rules (2 start + 2 end), got {len(rules)}" + assert len(markdowns) == 2, f"expected 2 Markdown sections, got {len(markdowns)}" + # Text "hi" went to current_text (buffered), not the transcript directly. + current_text.update.assert_any_call("hi") + # Transcript: [done] label + Markdown(response) (raw=False). log_writes = [_text_of(c[0][0]) for c in log.write.call_args_list] - assert "hi" in log_writes assert any(w.startswith("[done]") for w in log_writes if isinstance(w, str)) def test_render_exception_fallback(self) -> None: - """render_exception_fallback [adversarial, v0.5.0]: - widget.update raises → debug_log gets BOTH a plain-labeled fallback line - for the original Thinking event AND a `[render_error] ` - line (NO exception message per INV-009 security clause). transcript - receives nothing — routing preservation under failure (debug-shaped - event falls back to debug_log). + """render_exception_fallback [adversarial, v0.6.0]: + widget.update raises → thinking_log gets the plain-label fallback for + Thinking (per v0.6.0 routing — Thinking now routes to thinking_log, + not debug_log). render_error line follows. transcript untouched. """ from ratatoskr.tui import TuiPresenterState log = MagicMock() - debug_log = MagicMock() + thinking_log = MagicMock() widget = MagicMock() widget.update.side_effect = AttributeError("widget gone (msg should NOT leak)") state = TuiPresenterState() @@ -336,18 +325,15 @@ class TestTuiPresenterState: log=log, thinking_widget=widget, tools_log=MagicMock(), - debug_log=debug_log, + debug_log=MagicMock(), + current_text=MagicMock(), + thinking_log=thinking_log, raw=False, ) - writes = [c[0][0] for c in debug_log.write.call_args_list if isinstance(c[0][0], str)] - # POST-007: plain-label fallback for the original Thinking event. + writes = [c[0][0] for c in thinking_log.write.call_args_list if isinstance(c[0][0], str)] assert any(w.startswith("[thinking]") for w in writes), writes - # POST-007: render_error line with class name ONLY. assert any(w == "[render_error] AttributeError" for w in writes), writes - # Critical: exception message MUST NOT appear in any write (INV-009 security). assert not any("widget gone" in w for w in writes), writes - # v0.5.0 routing preservation: transcript receives NOTHING on a - # debug-shaped event's failure path. assert not log.write.called def test_state_reset_per_worker(self) -> None: @@ -361,7 +347,7 @@ class TestTuiPresenterState: thinking_widget=MagicMock(), tools_log=MagicMock(), debug_log=MagicMock(), - raw=False, + current_text=MagicMock(), thinking_log=MagicMock(), raw=False, ) s2 = TuiPresenterState() assert s1.thinking_open is True @@ -384,7 +370,7 @@ class TestTuiPresenterState: thinking_widget=widget, tools_log=MagicMock(), debug_log=debug_log, - raw=False, + current_text=MagicMock(), thinking_log=MagicMock(), raw=False, ) state.render( Cancelled( @@ -394,18 +380,19 @@ class TestTuiPresenterState: thinking_widget=widget, tools_log=MagicMock(), debug_log=debug_log, - raw=False, + current_text=MagicMock(), thinking_log=MagicMock(), raw=False, ) - # v0.5.0: closed thinking entry lands in debug_log; terminal [cancelled] in transcript. - debug_writes = [_text_of(c[0][0]) for c in debug_log.write.call_args_list] + # v0.6.0: closed thinking lands in thinking_log (Markdown body wrapped in + # Rule start/end). terminal [cancelled] still in transcript. log_writes = [_text_of(c[0][0]) for c in log.write.call_args_list] - assert any(w.startswith("· thinking: partial") for w in debug_writes) assert any(w.startswith("[cancelled]") for w in log_writes) assert widget.display is False def test_done_renders_markdown_after_label(self) -> None: - """done_renders_markdown_after_label [happy]: - Text("hi"), Done(response="hi") with raw=False → [done] label, Rule, Markdown in RichLog. + """done_renders_markdown_after_label [happy, v0.6.0]: + Text("hi") accumulates into current_text Static (buffered streaming); + Done(response="hi") with raw=False → [done] label + Rule + Markdown + in transcript. current_text cleared on terminal. """ from rich.markdown import Markdown from rich.rule import Rule @@ -413,6 +400,7 @@ class TestTuiPresenterState: from ratatoskr.tui import TuiPresenterState log = MagicMock() + current_text = MagicMock() widget = MagicMock() state = TuiPresenterState() state.render( @@ -421,22 +409,26 @@ class TestTuiPresenterState: thinking_widget=widget, tools_log=MagicMock(), debug_log=MagicMock(), + current_text=current_text, + thinking_log=MagicMock(), raw=False, ) + # Text accumulated to current_text, NOT written to log. + current_text.update.assert_any_call("hi") state.render( _make_tui_done(), log=log, thinking_widget=widget, tools_log=MagicMock(), debug_log=MagicMock(), + current_text=current_text, + thinking_log=MagicMock(), raw=False, ) + # Done cleared current_text and wrote [done] label + Rule + Markdown. + current_text.update.assert_any_call("") writes = [c[0][0] for c in log.write.call_args_list] - # Text stream wrote "hi" with no prefix. - assert "hi" in writes - # v0.5.1: [done] label is now RichText (Aurora green); plain content test via _text_of. assert any(_text_of(w).startswith("[done]") for w in writes) - # Rule + Markdown render present (post-Done body re-render per issue #4 INV-005). assert any(isinstance(w, Rule) for w in writes) assert any(isinstance(w, Markdown) for w in writes) @@ -456,7 +448,7 @@ class TestTuiPresenterState: thinking_widget=widget, tools_log=MagicMock(), debug_log=MagicMock(), - raw=True, + current_text=MagicMock(), thinking_log=MagicMock(), raw=True, ) state.render( _make_tui_done(), @@ -464,7 +456,7 @@ class TestTuiPresenterState: thinking_widget=widget, tools_log=MagicMock(), debug_log=MagicMock(), - raw=True, + current_text=MagicMock(), thinking_log=MagicMock(), raw=True, ) writes = [c[0][0] for c in log.write.call_args_list] assert not any(isinstance(w, Rule) for w in writes) @@ -488,7 +480,7 @@ class TestTuiPresenterState: thinking_widget=MagicMock(), tools_log=MagicMock(), debug_log=debug_log, - raw=False, + current_text=MagicMock(), thinking_log=MagicMock(), raw=False, ) # v0.5.0: WorkerPhase routes to debug_log, NOT transcript. assert not log.write.called @@ -527,7 +519,7 @@ class TestTuiPresenterState: thinking_widget=widget, tools_log=MagicMock(), debug_log=MagicMock(), - raw=True, + current_text=MagicMock(), thinking_log=MagicMock(), raw=True, ) # Belt-and-braces: widget cleared + hidden on EVERY terminal event. widget.update.assert_called_with("") @@ -551,7 +543,7 @@ class TestTuiPresenterState: thinking_widget=MagicMock(), tools_log=tools_log, debug_log=MagicMock(), - raw=False, + current_text=MagicMock(), thinking_log=MagicMock(), raw=False, ) # INV-014: write went to tools_log assert tools_log.write.called @@ -572,18 +564,21 @@ class TestTuiPresenterState: thinking_widget=MagicMock(), tools_log=tools_log, debug_log=MagicMock(), - raw=False, + current_text=MagicMock(), thinking_log=MagicMock(), raw=False, ) assert tools_log.write.called assert _text_of(tools_log.write.call_args[0][0]).startswith("· tool_result:") assert not log.write.called - def test_text_event_does_not_route_to_tools_log(self) -> None: - """text_event_does_not_route_to_tools_log [INV-015]: Text → transcript, NOT tools_log.""" + def test_text_event_buffers_into_current_text(self) -> None: + """text_event_buffers_into_current_text [v0.6.0]: Text → current_text Static + (accumulated), NOT log or tools_log. Streaming UX fix — no per-token spam. + """ from ratatoskr.tui import TuiPresenterState log = MagicMock() tools_log = MagicMock() + current_text = MagicMock() state = TuiPresenterState() state.render( Text(sse_id=SID, content="hello"), @@ -591,30 +586,35 @@ class TestTuiPresenterState: thinking_widget=MagicMock(), tools_log=tools_log, debug_log=MagicMock(), + current_text=current_text, + thinking_log=MagicMock(), raw=False, ) - assert log.write.called - assert log.write.call_args[0][0] == "hello" - # INV-015: tools_log was NOT written to + current_text.update.assert_called_once_with("hello") + assert not log.write.called assert not tools_log.write.called - def test_text_no_prefix(self) -> None: - """text_no_prefix [trace]: Text → RichLog line has no `·` prefix, no demotion.""" + def test_text_deltas_accumulate(self) -> None: + """text_deltas_accumulate [v0.6.0]: multiple Text deltas → current_text shows + concatenated content, NOT separate per-delta lines. + """ from ratatoskr.tui import TuiPresenterState - log = MagicMock() + current_text = MagicMock() state = TuiPresenterState() - state.render( - Text(sse_id=SID, content="hello"), - log=log, - thinking_widget=MagicMock(), - tools_log=MagicMock(), - debug_log=MagicMock(), - raw=False, - ) - line = log.write.call_args[0][0] - # Pure content, no demotion prefix. - assert line == "hello" + for tok in ("Hel", "lo", " ", "world"): + state.render( + Text(sse_id=SID, content=tok), + log=MagicMock(), + thinking_widget=MagicMock(), + tools_log=MagicMock(), + debug_log=MagicMock(), + current_text=current_text, + thinking_log=MagicMock(), + raw=False, + ) + # Final update reflects the full concatenation. + assert current_text.update.call_args_list[-1][0][0] == "Hello world" def test_duration_format_seconds(self) -> None: """duration_format_seconds [trace]: Done(duration_ms=5467) → label has "duration=5.5s".""" @@ -628,7 +628,7 @@ class TestTuiPresenterState: thinking_widget=MagicMock(), tools_log=MagicMock(), debug_log=MagicMock(), - raw=True, + current_text=MagicMock(), thinking_log=MagicMock(), raw=True, ) done_line = next( _text_of(c[0][0]) @@ -656,7 +656,7 @@ class TestTuiPresenterState: thinking_widget=MagicMock(), tools_log=MagicMock(), debug_log=MagicMock(), - raw=True, + current_text=MagicMock(), thinking_log=MagicMock(), raw=True, ) done_line = next( _text_of(c[0][0]) @@ -943,7 +943,7 @@ class TestLayoutShape: thinking_widget=app.query_one("#thinking-current"), tools_log=app.query_one("#tools-log", RichLog), debug_log=app.query_one("#debug-log", RichLog), - raw=True, + current_text=MagicMock(), thinking_log=MagicMock(), raw=True, ) done = next( c for c in seen @@ -973,7 +973,7 @@ class TestLayoutShape: await pilot.pause() debug_text = " ".join(str(line) for line in debug_log.lines) assert "no tool events" in tools_text - assert "waiting for telemetry" in debug_text + assert "worker_phase" in debug_text async def test_pane_name_updates_on_tab_switch(self) -> None: """pane_name_updates_on_tab_switch [v0.5.0]: pane-name reflects active tab. @@ -1161,7 +1161,10 @@ async def _submit_and_wait(app: RatatoskrApp, pilot, content: str) -> None: class TestStreamTurnWorker: @respx.mock async def test_happy_text_done_renders_markdown(self, monkeypatch: pytest.MonkeyPatch) -> None: - """happy_text_done_renders_markdown [happy,tracer]: …""" + """happy_text_done_renders_markdown [happy,tracer, v0.6.0]: + Text deltas go to current_text (not transcript); on Done, transcript + gets turn-header Rule, [done] label, post-Done Rule + Markdown body. + """ stream = _sse_chunk("42:1", {"type": "text", "content": "hello"}) + _sse_chunk( "42:2", _DONE_BODY ) @@ -1176,20 +1179,25 @@ class TestStreamTurnWorker: await pilot.pause() await _submit_and_wait(app, pilot, "hi") assert app.state == "idle" - # Streamed delta + done label + rule + markdown render - assert any(w == "hello" for w in writes) - assert any("[done]" in str(w) for w in writes) - # The post-Done markdown render uses rich Rule + Markdown — non-string writes. - # INV-005: BOTH separator (Rule) AND markdown render must be present in non-raw. + # v0.6.0: Text("hello") goes to current_text Static, NOT log. + # writes spy captures RichLog.write only, so "hello" SHOULD NOT appear. from rich.markdown import Markdown from rich.rule import Rule + assert not any(w == "hello" for w in writes) + assert any("[done]" in str(w) for w in writes) + # Post-Done: Markdown body + Rule + turn-header Rule all present. assert any(isinstance(w, Markdown) for w in writes) assert any(isinstance(w, Rule) for w in writes) @respx.mock async def test_raw_flag_skips_markdown_render(self, monkeypatch: pytest.MonkeyPatch) -> None: - """raw_flag_skips_markdown_render [trace]: …""" + """raw_flag_skips_markdown_render [trace, v0.6.0]: + With --raw, no Markdown render. A turn-header Rule IS still written + (v0.6.0 INV — turn correlation lives in every pane). The post-Done + Rule(separator) is suppressed; accumulated streamed text is written + as a plain string instead. + """ stream = _sse_chunk("42:1", {"type": "text", "content": "hi"}) + _sse_chunk( "42:2", _DONE_BODY ) @@ -1201,12 +1209,17 @@ class TestStreamTurnWorker: async with app.run_test() as pilot: await pilot.pause() await _submit_and_wait(app, pilot, "x") - # INV-005: with --raw, NEITHER Rule separator NOR Markdown render appears. from rich.markdown import Markdown from rich.rule import Rule + # No Markdown in raw mode. assert not any(isinstance(w, Markdown) for w in writes) - assert not any(isinstance(w, Rule) for w in writes) + # Only turn-header Rules — one per pane (transcript + tools + + # debug + thinking = 4). No post-Done separator Rule. + rules = [w for w in writes if isinstance(w, Rule)] + assert len(rules) == 4, f"expected 4 turn-header Rules, got {len(rules)}" + # Accumulated text "hi" written as plain string post-Done. + assert "hi" in writes @respx.mock async def test_error_terminal_returns_to_idle(self, monkeypatch: pytest.MonkeyPatch) -> None: diff --git a/uv.lock b/uv.lock index b0b29d6..9622592 100644 --- a/uv.lock +++ b/uv.lock @@ -968,7 +968,7 @@ wheels = [ [[package]] name = "ratatoskr" -version = "0.5.1" +version = "0.6.0" source = { editable = "." } dependencies = [ { name = "httpx" },