Compare commits

...

2 Commits

Author SHA1 Message Date
vh 209427ab23 feat(tui): debug-pane audit logging surface (v0.10.0)
Adds wire-level visibility appropriate for a debugging TUI. Every
SSE event arrival now lands as one debug-pane line; token-rate Text
and Thinking deltas get aggregated counters surfaced in a per-turn
summary instead of per-delta spam.

Audit surfaces added (all routed to the debug pane):
- per-event arrival: timestamp + event type + sse_id + event-specific
  summary for WorkerPhase / ToolStart / ToolResult / TextBoundary /
  Done / Error / Cancelled
- turn-summary at terminal events: text_deltas / text_bytes /
  thinking_deltas / thinking_bytes / elapsed_ms
- app-level state-machine transitions via new RatatoskrApp._transition
  helper (idle → streaming → cancelling → idle, with reason)
- worker_spawn line at on_input_submitted with content_len
- ctrl_c / ctrl_d audit lines documenting action + exit code
- cancel POST lifecycle: _cancel_via_sse takes an optional audit
  callback and emits issued / ok / failed lines
- app_mounted bootstrap line at on_mount (server + agent + session
  tail + raw + end_user_id)
- wire-error exception class + body audit at _stream_turn_worker

Helpers:
- TuiPresenterState: text_delta_count / text_byte_count /
  thinking_delta_count / thinking_byte_count / turn_start_ts
- module-level _ts() + _audit_line() + RatatoskrApp._audit() /
  _transition()

Tests: 6 new test cases lock in audit-line shape, turn-summary
aggregation, cancel-POST lifecycle callback, and the silence of
per-Text-delta debug writes.
2026-05-25 01:36:35 -07:00
vh 139771c8d8 feat(tui): live Markdown rendering during text streaming (v0.9.0)
Replaces v0.8.2's drop-Markdown patch with proper in-place Markdown
rendering. The transcript becomes a VerticalScroll container; each
turn's response body lives as a single Static widget whose content
is updated as Text deltas arrive — Markdown is re-rendered in place
rather than re-printed on Done. Eliminates the v0.8.x double-print
without sacrificing rich formatting.

- transcript: RichLog → VerticalScroll (#transcript-scroll)
- Text deltas: mount Static(Markdown(buffer)) on first delta;
  Static.update(Markdown(buffer)) on subsequent deltas
- --raw mode: bypass Markdown, mount Static(plain_str) for the same
  in-place update semantics
- Terminal events (Done/Error/Cancelled) mount styled label Statics
- _cancel_via_sse: write → mount Static on the new container
- _write_turn_headers: transcript gets a styled RichText Static
  ("── turn N ──"); other panes still receive Rule renderables
- Test suite reshape: bulk rename `log` → `transcript` for the
  presenter contract, `_mounted_renderables` helper extracts
  Static.content for assertion, `_spy_writes` captures both
  RichLog.write and VerticalScroll.mount
2026-05-24 22:18:45 -07:00
4 changed files with 750 additions and 241 deletions
+1 -1
View File
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
[project]
name = "ratatoskr"
version = "0.8.2"
version = "0.10.0"
description = "Worldtree Conversation API debug TUI — multi-pane observability dashboard"
readme = "README.md"
requires-python = ">=3.12"
+360 -91
View File
@@ -10,13 +10,15 @@ from __future__ import annotations
import asyncio
import sys
import time as _time
from dataclasses import dataclass
from datetime import datetime as _datetime
from typing import ClassVar, Literal
import httpx
from textual.app import App, ComposeResult
from textual.binding import Binding
from textual.containers import Horizontal, Vertical
from textual.containers import Horizontal, Vertical, VerticalScroll
from textual.theme import Theme
from textual.widgets import (
Footer,
@@ -178,6 +180,48 @@ def _plain_label(event: Event) -> str:
return f"[unknown_event] {type(event).__name__}"
def _ts() -> str:
"""HH:MM:SS.fff wall-clock timestamp for debug-pane log lines."""
now = _datetime.now()
return now.strftime("%H:%M:%S") + f".{now.microsecond // 1000:03d}"
def _audit_line(event: Event) -> str:
"""One-line wire-level audit summary for the debug pane.
v0.10.0: every SSE event arrival lands as one of these in the debug
pane (Text and Thinking deltas are aggregated into the turn summary
instead — token-rate per-delta lines would drown the pane). Shape:
`[HH:MM:SS.fff] event_type sse_id=T:S key=val …`.
"""
sid = getattr(event, "sse_id", None)
sid_str = f"{sid.turn_id}:{sid.seq}" if sid is not None else "-"
kind = type(event).__name__.lower()
if isinstance(event, WorkerPhase):
detail = f"phase={event.phase} turn_id={event.turn_id}"
elif isinstance(event, ToolStart):
detail = f"name={event.name} args={event.arguments!r:.80}"
elif isinstance(event, ToolResult):
detail = f"name={event.name} duration_ms={event.duration_ms}"
elif isinstance(event, TextBoundary):
detail = f"kind={event.kind} char_offset={event.char_offset}"
elif isinstance(event, Done):
detail = (
f"turn_id={event.sse_id.turn_id} model={event.model} "
f"duration_ms={event.duration_ms}"
)
elif isinstance(event, Error):
detail = (
f"turn_id={event.sse_id.turn_id} code={event.error_code} "
f"message={event.message!r:.80}"
)
elif isinstance(event, Cancelled):
detail = f"turn_id={event.turn_id} reason={event.reason!r}"
else: # Text / Thinking handled by counter path; fallback for safety
detail = ""
return f"[{_ts()}] {kind} sse_id={sid_str} {detail}".rstrip()
@dataclass(slots=True)
class TuiPresenterState:
"""Per-turn presenter state for TUI mode (issue #12).
@@ -194,20 +238,29 @@ class TuiPresenterState:
# only on `\n` boundaries (one written line per natural paragraph) or
# when the run closes (any leftover tail).
thinking_chunk_buffer: str = ""
# v0.8.1: same pattern for Text deltas. Pre-v0.8.1 the Text deltas
# streamed into a dedicated #current-text Static below the transcript;
# that Static (docked-bottom, height: auto) grew during streaming and
# visually OVERLAPPED the transcript above (Textual didn't dynamically
# resize the 1fr transcript while the dock-bottom child expanded).
# The Static is gone in v0.8.1 — Text deltas coalesce on `\n` and write
# directly to `log` (transcript), the same shape thinking uses.
# v0.9.0: Text accumulator for live Markdown rendering. Worldtree emits
# Text deltas at token granularity; each delta appends to this buffer
# and the current_response_widget re-renders Markdown(text_chunk_buffer)
# in place. On terminal event the widget is finalized + reference clears.
text_chunk_buffer: str = ""
# v0.9.0: reference to the Static widget holding the current turn's
# response Markdown Renderable. None between turns.
current_response_widget: object = None
# v0.10.0: per-turn counters for the debug-pane turn-summary line. Text
# and Thinking events arrive at token rate; emitting per-delta debug
# lines would drown the pane. Instead we count them and surface
# aggregated totals when the turn closes.
text_delta_count: int = 0
text_byte_count: int = 0
thinking_delta_count: int = 0
thinking_byte_count: int = 0
turn_start_ts: float = 0.0
def render(
self,
event: Event,
*,
log: RichLog,
transcript: "VerticalScroll",
tools_log: RichLog,
debug_log: RichLog,
thinking_log: RichLog,
@@ -215,14 +268,14 @@ class TuiPresenterState:
) -> None:
"""Render one Worldtree SSE event with the TUI hierarchy + coalescing.
v0.8.1 routing:
- `log` (transcript) = chat content: user-prompt echo (written
outside the presenter), coalesced Text deltas, terminal labels,
optional post-Done Markdown body.
- `tools_log` = ToolStart + ToolResult.
- `debug_log` = WorkerPhase + TextBoundary.
- `thinking_log` = streaming Thinking deltas inline (coalesced on
`\n`). Rule(start)/Rule(end) wrap each run.
v0.9.0 routing:
- `transcript` (VerticalScroll) = chat content: each turn mounts
child widgets (turn-header / prompt-echo / response Markdown /
done-label). Live Markdown rendering during Text streaming.
- `tools_log` (RichLog) = ToolStart + ToolResult.
- `debug_log` (RichLog) = WorkerPhase + TextBoundary.
- `thinking_log` (RichLog) = streaming Thinking deltas inline
(coalesced on `\n`); Rule(start)/Rule(end) wrap each run.
Exceptions caught at the presenter boundary (INV-009 fallback).
"""
@@ -240,6 +293,29 @@ class TuiPresenterState:
return RichText(s, style=_AU_DEMOTED)
try:
# v0.10.0: per-event audit log line to debug pane. Text and
# Thinking arrive at token rate, so we count them rather than
# emit a line per delta — totals are reported in the turn-
# summary on Done/Error/Cancelled. Everything else gets one
# debug-pane line per arrival with timestamp + sse_id + a short
# event-specific summary, giving the operator a wire-level
# timeline of what the server sent.
if isinstance(event, Text):
if self.text_delta_count == 0:
if self.turn_start_ts == 0.0:
self.turn_start_ts = _time.monotonic()
self.text_delta_count += 1
self.text_byte_count += len(event.content)
elif isinstance(event, Thinking):
if self.thinking_delta_count == 0:
if self.turn_start_ts == 0.0:
self.turn_start_ts = _time.monotonic()
self.thinking_delta_count += 1
self.thinking_byte_count += len(event.content)
else:
if self.turn_start_ts == 0.0:
self.turn_start_ts = _time.monotonic()
debug_log.write(_dim(_audit_line(event)))
# v0.7.1: Thinking deltas coalesce by newline before flushing.
# Worldtree emits Thinking events at token granularity; per-delta
# RichLog writes produce one visual line per token (per-token-per-
@@ -288,47 +364,95 @@ class TuiPresenterState:
# coalesced on `\n`. Same pattern as Thinking (v0.7.1).
# The pre-v0.8.1 #current-text Static is gone — its dock-
# bottom growth was overlapping the transcript visually.
#
# v0.9.0: Text deltas accumulate in text_chunk_buffer and
# the current_response_widget renders Markdown(buffer) in
# place. First Text delta of the turn mounts a fresh Static
# holding the Markdown Renderable; subsequent deltas update
# the same widget. Live markdown rendering — no post-Done
# re-render needed.
from rich.markdown import Markdown
self.text_chunk_buffer += event.content
while "\n" in self.text_chunk_buffer:
line, _, rest = self.text_chunk_buffer.partition("\n")
if line:
log.write(line)
self.text_chunk_buffer = rest
# --raw bypasses Markdown rendering — useful for debugging
# the raw text stream surface, and matches the pre-v0.9.0
# --raw semantics (which dropped the post-Done Markdown re-
# render). In raw mode the response widget holds plain str.
rendered = (
self.text_chunk_buffer if raw else Markdown(self.text_chunk_buffer)
)
if self.current_response_widget is None:
self.current_response_widget = Static(
rendered, classes="response-md"
)
transcript.mount(self.current_response_widget)
else:
self.current_response_widget.update(rendered)
transcript.scroll_end(animate=False)
return
if isinstance(event, (Done, Error, Cancelled)):
# Terminal event: flush any remaining text tail before the
# label / Markdown body lands.
if self.text_chunk_buffer:
log.write(self.text_chunk_buffer)
self.text_chunk_buffer = ""
# Terminal labels tinted per outcome (Aurora green / Dawn red
# / Dawn yellow) for at-a-glance scanning.
# v0.10.0: emit turn-summary to debug pane before clearing
# counters. Aggregates the per-event totals (Text + Thinking
# deltas don't get per-event audit lines because they arrive
# at token rate; the summary surfaces what was elided).
elapsed_ms = (
int((_time.monotonic() - self.turn_start_ts) * 1000)
if self.turn_start_ts
else 0
)
turn_id = (
event.sse_id.turn_id
if hasattr(event, "sse_id")
else getattr(event, "turn_id", "?")
)
debug_log.write(_dim(
f"[{_ts()}] turn_summary turn_id={turn_id} "
f"text_deltas={self.text_delta_count} "
f"text_bytes={self.text_byte_count} "
f"thinking_deltas={self.thinking_delta_count} "
f"thinking_bytes={self.thinking_byte_count} "
f"elapsed_ms={elapsed_ms}"
))
# Terminal event: finalize the response widget (clear ref so
# the next turn mounts a fresh one). The accumulated text is
# already rendered as Markdown in the widget — no post-Done
# re-render, no double-print.
self.text_chunk_buffer = ""
self.current_response_widget = None
# Terminal labels mount as styled Statics. Tinted per outcome
# (Aurora green / Dawn red / Dawn yellow) for at-a-glance
# scanning.
if isinstance(event, Done):
log.write(RichText(
f"[done] turn_id={event.sse_id.turn_id} model={event.model} "
f"duration={_format_duration_ms(event.duration_ms)} "
f"usage {_format_usage(event.usage, arrow='')}",
style=_AU_SUCCESS,
transcript.mount(Static(
RichText(
f"[done] turn_id={event.sse_id.turn_id} "
f"model={event.model} "
f"duration={_format_duration_ms(event.duration_ms)} "
f"usage {_format_usage(event.usage, arrow='')}",
style=_AU_SUCCESS,
),
classes="done-label",
))
# v0.8.2: post-Done Markdown body re-render dropped. Pre-
# v0.8.2 the transcript got BOTH the streamed text AND
# the Markdown(response) re-render — same content twice,
# operator-flagged as "double prints". The streamed text
# IS the response now; markdown formatting (bold, lists,
# code) renders as plain text. Matches thinking pane's
# stream-as-content semantics (no post-close re-render).
elif isinstance(event, Error):
log.write(RichText(
f"[error] turn_id={event.sse_id.turn_id} code={event.error_code} "
f"message={event.message!r}",
style=_AU_ERROR,
transcript.mount(Static(
RichText(
f"[error] turn_id={event.sse_id.turn_id} "
f"code={event.error_code} message={event.message!r}",
style=_AU_ERROR,
),
classes="error-label",
))
else: # Cancelled
log.write(RichText(
f"[cancelled] turn_id={event.turn_id} reason={event.reason!r} "
f"partial_message_id={event.partial_message_id}",
style=_AU_WARNING,
transcript.mount(Static(
RichText(
f"[cancelled] turn_id={event.turn_id} "
f"reason={event.reason!r} "
f"partial_message_id={event.partial_message_id}",
style=_AU_WARNING,
),
classes="cancelled-label",
))
transcript.scroll_end(animate=False)
return
if isinstance(event, WorkerPhase):
# v0.5.0: telemetry → Debug pane, not transcript.
@@ -360,22 +484,24 @@ class TuiPresenterState:
# the original event AND a render_error line with the class name only
# (NO exception message — security clause). Volva F1 fix.
#
# v0.6.0 routing-under-failure preservation — fallback writes go
# to the same destination the successful render would have used:
# - ToolStart/ToolResult → tools_log
# - Thinking → thinking_log
# - WorkerPhase/TextBoundary → debug_log
# - everything else → log
# v0.9.0 routing-under-failure: panes (RichLog) still write Strip
# lines; transcript (VerticalScroll) mounts a Static instead.
if isinstance(event, (ToolStart, ToolResult)):
target = tools_log
tools_log.write(_plain_label(event))
tools_log.write(f"[render_error] {type(exc).__name__}")
elif isinstance(event, Thinking):
target = thinking_log
thinking_log.write(_plain_label(event))
thinking_log.write(f"[render_error] {type(exc).__name__}")
elif isinstance(event, (WorkerPhase, TextBoundary)):
target = debug_log
debug_log.write(_plain_label(event))
debug_log.write(f"[render_error] {type(exc).__name__}")
else:
target = log
target.write(_plain_label(event))
target.write(f"[render_error] {type(exc).__name__}")
# Transcript-bound event (Text / Done / Error / Cancelled).
transcript.mount(Static(_plain_label(event), classes="error-label"))
transcript.mount(
Static(f"[render_error] {type(exc).__name__}", classes="error-label")
)
transcript.scroll_end(animate=False)
class AgentPickerApp(App[str | None]):
@@ -564,11 +690,40 @@ class RatatoskrApp(App[int]):
}
/* v0.6.5: thinking-current Static removed; thinking now streams
directly into thinking-log so the whole pane scrolls naturally. */
#transcript {
/* v0.9.0: transcript is a VerticalScroll container holding dynamically
mounted Statics + Markdown widgets per turn. Live Markdown rendering
replaces the v0.8.x RichLog approach which couldn't render Markdown
in-flight (only on Done as a re-render → double-print bug). */
#transcript-scroll {
height: 1fr;
background: $background;
padding: 0 1;
}
/* Per-turn mounted widgets carry id-prefix conventions:
- .turn-header "── turn N ──" (dim)
- .prompt-echo " user input" (aurora bright cyan)
- .response-md Markdown(accumulated_text) — updated live
- .done-label "[done] turn_id=…" (aurora green)
- .error-label "[error] …" (dawn red)
- .cancelled-label "[cancelled] …" (dawn yellow)
*/
.turn-header {
height: auto;
padding: 0 1;
color: $au-dark-60;
}
.prompt-echo {
height: auto;
padding: 0 1;
}
.response-md {
height: auto;
padding: 0 1;
}
.done-label, .error-label, .cancelled-label {
height: auto;
padding: 0 1;
}
/* v0.8.1: #current-text Static removed. Streaming text now coalesces
on `\n` and writes directly to #transcript (same pattern as v0.7.1
thinking fix). Eliminates the dock-bottom-growth-overlap bug. */
@@ -670,7 +825,12 @@ class RatatoskrApp(App[int]):
# work without widget-level markup=True.
with Horizontal(id="main-row"):
with Vertical(id="left-column"):
yield RichLog(id="transcript", wrap=True, markup=False, highlight=False)
# v0.9.0: transcript is a VerticalScroll holding per-turn
# mounted widgets (turn header, prompt echo, response Markdown,
# done label). Live Markdown rendering happens via Static
# widgets holding `Markdown` Renderables, updated as Text
# deltas arrive.
yield VerticalScroll(id="transcript-scroll")
yield Input(id="prompt", placeholder="Type a message and press Enter")
with Vertical(id="right-column"):
with TabbedContent(id="side-panes"):
@@ -731,22 +891,40 @@ class RatatoskrApp(App[int]):
)
self.state = "idle"
self._set_hint(self.HINT_IDLE)
# v0.10.0: startup audit so the debug pane carries a complete
# session bootstrap line (server URL, agent, end_user_id, raw flag,
# session tail) before the first turn fires.
self._audit(
f"app_mounted server={self.args.server_url} agent_id={self.agent_id!r} "
f"session={self.session_id[-8:]} raw={self.args.raw} "
f"end_user_id={getattr(self.args, 'end_user_id', None)!r}"
)
def _write_turn_headers(self, turn_id: int) -> None:
"""v0.6.0: Write `── turn N ──` Rule headers across every pane so
operators can visually correlate sections during cross-pane
debugging. Called from `_stream_turn_worker` on first event of
each new turn (idempotent per turn via active_turn_id guard).
"""v0.6.0: turn-ID headers across every pane for cross-pane
correlation. v0.9.0: transcript is a VerticalScroll; mounts a
Static with rule-style text instead of writing a Rule Renderable
to RichLog. Other panes still use RichLog.write(Rule).
"""
from rich.rule import Rule
from rich.text import Text as RichText
title = f"turn {turn_id}"
rule = Rule(title=title, style=_AU_DEMOTED)
try:
self.query_one("#transcript", RichLog).write(rule)
# Transcript (VerticalScroll): mount a styled Static.
transcript = self.query_one("#transcript-scroll", VerticalScroll)
transcript.mount(
Static(
RichText(f"── turn {turn_id} ──", style=_AU_DEMOTED),
classes="turn-header",
)
)
# Other panes (RichLog): write the Rule Renderable.
self.query_one("#tools-log", RichLog).write(rule)
self.query_one("#debug-log", RichLog).write(rule)
self.query_one("#thinking-log", RichLog).write(rule)
transcript.scroll_end(animate=False)
except Exception:
# Defensive: widget tree may be tearing down — never let a
# turn-header write block the SSE consumer.
@@ -761,13 +939,51 @@ class RatatoskrApp(App[int]):
# Widget may be gone during shutdown; ignore.
pass
def _audit(self, line: str) -> None:
"""Write a timestamped audit line to the debug pane.
v0.10.0: shared sink for app-level events that don't pass through
the presenter — state transitions, worker spawn/cancel, cancel POST
lifecycle, startup probes. The presenter's per-event audit lives at
`_audit_line()`; this is its app-side counterpart.
"""
try:
from rich.text import Text as RichText
self.query_one("#debug-log", RichLog).write(
RichText(f"[{_ts()}] {line}", style=_AU_DEMOTED)
)
except Exception:
# Widget may not exist yet (pre-mount) or be tearing down.
pass
def _transition(
self, new_state: Literal["idle", "streaming", "cancelling"], reason: str
) -> None:
"""Set self.state with debug-pane audit log.
Every state machine transition flows through here so the debug pane
carries a complete idle→streaming→cancelling→idle timeline with the
triggering reason. Cheap; safe to call from any context.
"""
old = self.state
self.state = new_state
if old != new_state:
self._audit(f"state {old}{new_state} reason={reason}")
async def on_input_submitted(self, event: Input.Submitted) -> None:
"""Echo user prompt, spawn stream worker; busy notice if not idle."""
"""Echo user prompt, spawn stream worker; busy notice if not idle.
v0.9.0: prompt echo mounts as a Static in the transcript VerticalScroll
(was log.write to RichLog).
"""
if event.input.id != "prompt":
return
log = self.query_one("#transcript", RichLog)
transcript = self.query_one("#transcript-scroll", VerticalScroll)
if self.state != "idle":
log.write("[busy] turn in flight; input ignored")
transcript.mount(
Static("[busy] turn in flight; input ignored", classes="error-label")
)
transcript.scroll_end(animate=False)
event.input.value = ""
return
content = event.input.value.strip()
@@ -776,35 +992,53 @@ class RatatoskrApp(App[int]):
# v0.4.1 retheme: operator's voice gets Australis bright cyan so it
# stands out against the default-foreground assistant text below it.
from rich.text import Text as RichText
log.write(RichText(f" {content}", style=_AU_USER_ECHO)) # noqa: RUF001
transcript.mount(
Static(
RichText(f" {content}", style=_AU_USER_ECHO), # noqa: RUF001
classes="prompt-echo",
)
)
transcript.scroll_end(animate=False)
event.input.value = ""
self.state = "streaming"
self._transition("streaming", "input_submitted")
self._audit(f"worker_spawn content_len={len(content)}")
self._set_hint(self.HINT_STREAMING)
self.stream_worker = self.run_worker(
self._stream_turn_worker(content), exclusive=True
)
async def _stream_turn_worker(self, content: str) -> None:
"""Drive stream_turn, render events via TuiPresenterState (issue #12)."""
"""Drive stream_turn, render events via TuiPresenterState.
v0.9.0: transcript is a VerticalScroll; the presenter's `transcript`
argument is the container, and the presenter mounts Static / Markdown-
backed widgets directly. Wire-error labels mount as `error-label`
Statics into the transcript-scroll.
"""
assert self.state == "streaming"
assert self.client is not None
assert content
log = self.query_one("#transcript", RichLog)
transcript = self.query_one("#transcript-scroll", VerticalScroll)
tools_log = self.query_one("#tools-log", RichLog)
debug_log = self.query_one("#debug-log", RichLog)
thinking_log = self.query_one("#thinking-log", RichLog)
presenter = TuiPresenterState()
def _mount_wire_error(label: str) -> None:
try:
transcript.mount(Static(label, classes="error-label"))
transcript.scroll_end(animate=False)
except Exception:
pass
try:
async for event in stream_turn(self.client, self.session_id, content):
if self.active_turn_id is None:
self.active_turn_id = event.sse_id.turn_id
# v0.6.0: turn-ID headers across all panes so the
# operator can visually correlate sections during
# cross-pane debugging.
self._write_turn_headers(self.active_turn_id)
presenter.render(
event,
log=log,
transcript=transcript,
tools_log=tools_log,
debug_log=debug_log,
thinking_log=thinking_log,
@@ -813,17 +1047,22 @@ class RatatoskrApp(App[int]):
if isinstance(event, (Done, Error, Cancelled)):
break
except SseConnectFailed as exc:
log.write(f"[sse_connect_failed] status={exc.status} body={exc.body!r}")
self._audit(f"sse_connect_failed status={exc.status} body={exc.body!r:.120}")
_mount_wire_error(f"[sse_connect_failed] status={exc.status} body={exc.body!r}")
except SseConnectionDropped as exc:
log.write(f"[connection_dropped] last_seen={exc.last_seen_sse_id}")
self._audit(f"connection_dropped last_seen={exc.last_seen_sse_id}")
_mount_wire_error(f"[connection_dropped] last_seen={exc.last_seen_sse_id}")
except MalformedSseId as exc:
log.write(f"[malformed_sse_id] raw={exc.raw!r}")
self._audit(f"malformed_sse_id raw={exc.raw!r}")
_mount_wire_error(f"[malformed_sse_id] raw={exc.raw!r}")
except MalformedSseData as exc:
log.write(f"[malformed_sse_data] raw={exc.raw!r}")
self._audit(f"malformed_sse_data raw={exc.raw!r:.120}")
_mount_wire_error(f"[malformed_sse_data] raw={exc.raw!r}")
except TurnIdFlip as exc:
log.write(f"[turn_id_flip] expected={exc.established} got={exc.got}")
self._audit(f"turn_id_flip expected={exc.established} got={exc.got}")
_mount_wire_error(f"[turn_id_flip] expected={exc.established} got={exc.got}")
finally:
self.state = "idle"
self._transition("idle", "worker_finally")
self.active_turn_id = None
self._set_hint(self.HINT_IDLE)
@@ -835,26 +1074,35 @@ class RatatoskrApp(App[int]):
"""Two-stage Ctrl-C state machine per INV-003."""
assert self.state in ("idle", "streaming", "cancelling")
if self.state == "idle":
self._audit("ctrl_c state=idle action=exit code=0")
self.exit(0)
elif self.state == "streaming":
if self.active_turn_id is None:
self._audit("ctrl_c state=streaming active_turn_id=None action=force_exit code=3")
if self.stream_worker is not None:
self.stream_worker.cancel()
self.exit(3)
return
self.state = "cancelling"
self._audit(f"ctrl_c state=streaming turn_id={self.active_turn_id} action=cancel_post")
self._transition("cancelling", "ctrl_c_cancel_post_issued")
self._set_hint(self.HINT_CANCELLING)
log = self.query_one("#transcript", RichLog)
transcript = self.query_one("#transcript-scroll", VerticalScroll)
self.run_worker(
_cancel_via_sse(self.client, self.session_id, self.active_turn_id, log=log)
_cancel_via_sse(
self.client, self.session_id, self.active_turn_id,
transcript=transcript,
audit=self._audit,
)
)
elif self.state == "cancelling":
self._audit("ctrl_c state=cancelling action=force_exit code=3")
if self.stream_worker is not None:
self.stream_worker.cancel()
self.exit(3)
def action_quit(self) -> None:
"""Ctrl-D — immediate exit regardless of state."""
self._audit(f"ctrl_d state={self.state} action=exit code=0")
if self.stream_worker is not None and not self.stream_worker.is_finished:
self.stream_worker.cancel()
self.exit(0)
@@ -995,12 +1243,33 @@ async def _cancel_via_sse(
session_id: str,
turn_id: int,
*,
log: RichLog,
transcript: VerticalScroll,
audit: "Callable[[str], None] | None" = None,
) -> None:
"""Fire-and-forget cancel; never raises (mirrors cli._cancel_and_log; #3 INV-009)."""
"""Fire-and-forget cancel; never raises (mirrors cli._cancel_and_log; #3 INV-009).
v0.9.0: mounts a `[cancel_failed]` Static into the transcript-scroll
container on failure (was log.write to RichLog).
v0.10.0: optional `audit` callback (RatatoskrApp._audit) receives one
line on POST issue + one on POST result, so the debug pane carries the
full cancel lifecycle. Defaults to no-op for legacy callers.
"""
assert client is not None
assert isinstance(turn_id, int) and turn_id > 0
if audit is not None:
audit(f"cancel_post issued session_id={session_id} turn_id={turn_id}")
try:
await cancel_turn(client, session_id, turn_id)
if audit is not None:
audit(f"cancel_post ok turn_id={turn_id}")
except (CancelFailed, CancelTurnNotFound, CancelAlreadyCompleted, httpx.RequestError) as exc:
log.write(f"[cancel_failed] {type(exc).__name__}: {exc}")
if audit is not None:
audit(f"cancel_post failed turn_id={turn_id} {type(exc).__name__}: {exc!s:.120}")
try:
transcript.mount(Static(
f"[cancel_failed] {type(exc).__name__}: {exc}",
classes="error-label",
))
transcript.scroll_end(animate=False)
except Exception:
pass
+388 -148
View File
@@ -61,20 +61,43 @@ def _args_existing(session_id: str = "s-1existing", **overrides) -> ParsedArgs:
def _spy_writes(monkeypatch) -> list:
"""Patch RichLog.write to record every arg into a list (returned).
"""Patch RichLog.write AND VerticalScroll.mount to record every renderable
or mounted-widget content into a single list (returned).
Accepts *args/**kwargs so Textual's internal deferred-render path
(which calls write positionally with width/expand/shrink/scroll_end)
still works after a write-during-mount + Resize sequence.
v0.9.0: transcript content is mounted into a VerticalScroll, not written
to a RichLog. The spy captures both shapes — for each mounted Static, the
Static's `renderable` (Markdown / RichText / str) lands in the list,
indistinguishably from RichLog.write entries. Integration tests assert
on substrings or types in `writes` so the merged shape is the right
abstraction.
Accepts *args/**kwargs so Textual's internal deferred-render paths still
work after a write-during-mount + Resize sequence.
"""
from textual.containers import VerticalScroll
from textual.widgets import Static
writes: list = []
original = RichLog.write
def spy(self, content, *args, **kw):
original_write = RichLog.write
def spy_write(self, content, *args, **kw):
writes.append(content)
return original(self, content, *args, **kw)
return original_write(self, content, *args, **kw)
monkeypatch.setattr(RichLog, "write", spy)
monkeypatch.setattr(RichLog, "write", spy_write)
original_mount = VerticalScroll.mount
def spy_mount(self, *children, **kw):
for child in children:
if isinstance(child, Static):
writes.append(child.content)
else:
writes.append(child)
return original_mount(self, *children, **kw)
monkeypatch.setattr(VerticalScroll, "mount", spy_mount)
return writes
@@ -126,13 +149,13 @@ class TestTuiPresenterState:
from ratatoskr.tui import TuiPresenterState
log = MagicMock()
transcript = MagicMock()
thinking_log = MagicMock()
state = TuiPresenterState()
for chunk in ("Let", " me", " think"):
state.render(
Thinking(sse_id=SID, content=chunk),
log=log,
transcript=transcript,
tools_log=MagicMock(),
debug_log=MagicMock(),
thinking_log=thinking_log,
@@ -143,7 +166,7 @@ class TestTuiPresenterState:
assert len(writes) == 1
assert isinstance(writes[0], Rule)
assert state.thinking_chunk_buffer == "Let me think"
assert log.write.call_count == 0
assert transcript.mount.call_count == 0
def test_thinking_flushes_on_newline(self) -> None:
"""thinking_flushes_on_newline [happy, v0.7.1]:
@@ -156,7 +179,7 @@ class TestTuiPresenterState:
for chunk in ("Hello", " world", "\n"):
state.render(
Thinking(sse_id=SID, content=chunk),
log=MagicMock(),
transcript=MagicMock(),
tools_log=MagicMock(),
debug_log=MagicMock(),
thinking_log=thinking_log,
@@ -178,14 +201,14 @@ class TestTuiPresenterState:
from ratatoskr.tui import TuiPresenterState
log = MagicMock()
transcript = MagicMock()
debug_log = MagicMock()
thinking_log = MagicMock()
state = TuiPresenterState()
for content in ("a", "b"):
state.render(
Thinking(sse_id=SID, content=content),
log=log,
transcript=transcript,
tools_log=MagicMock(),
debug_log=debug_log,
thinking_log=thinking_log,
@@ -193,7 +216,7 @@ class TestTuiPresenterState:
)
state.render(
WorkerPhase(sse_id=SID, phase="streaming", turn_id=42),
log=log,
transcript=transcript,
tools_log=MagicMock(),
debug_log=debug_log,
thinking_log=thinking_log,
@@ -207,7 +230,7 @@ class TestTuiPresenterState:
assert isinstance(thinking_writes[2], Rule)
# worker_phase still goes to debug_log; transcript untouched.
assert "· worker_phase:" in _text_of(debug_log.write.call_args_list[-1][0][0])
assert not log.write.called
assert not transcript.mount.called
# v0.6.5: thinking-current Static removed; test_thinking_widget_truncation
# and test_thinking_widget_visibility_lifecycle deleted (no longer apply).
@@ -224,7 +247,7 @@ class TestTuiPresenterState:
from ratatoskr.tui import TuiPresenterState
log = MagicMock()
transcript = MagicMock()
thinking_log = MagicMock()
state = TuiPresenterState()
for evt in (
@@ -233,13 +256,13 @@ class TestTuiPresenterState:
Thinking(sse_id=SID, content="second"),
):
state.render(
evt, log=log,
evt, transcript=transcript,
tools_log=MagicMock(), debug_log=MagicMock(),
thinking_log=thinking_log, raw=False,
)
state.render(
_make_tui_done(),
log=log,
transcript=transcript,
tools_log=MagicMock(), debug_log=MagicMock(),
thinking_log=thinking_log, raw=False,
)
@@ -251,9 +274,9 @@ class TestTuiPresenterState:
assert "first" in delta_strs
assert "second" in delta_strs
# v0.8.1: Text "hi" flushes as a line in transcript on Done.
log_writes = [_text_of(c[0][0]) for c in log.write.call_args_list]
assert "hi" in log_writes
assert any(w.startswith("[done]") for w in log_writes if isinstance(w, str))
transcript_renderables = [_text_of(r) for r in _mounted_renderables(transcript)]
assert "hi" in transcript_renderables
assert any(w.startswith("[done]") for w in transcript_renderables if isinstance(w, str))
def test_render_exception_fallback(self) -> None:
"""render_exception_fallback [adversarial, v0.6.5]:
@@ -263,7 +286,7 @@ class TestTuiPresenterState:
"""
from ratatoskr.tui import TuiPresenterState
log = MagicMock()
transcript = MagicMock()
thinking_log = MagicMock()
# First call (Rule write) raises; subsequent calls succeed for fallback.
thinking_log.write.side_effect = [
@@ -274,7 +297,7 @@ class TestTuiPresenterState:
state = TuiPresenterState()
state.render(
Thinking(sse_id=SID, content="x"),
log=log,
transcript=transcript,
tools_log=MagicMock(),
debug_log=MagicMock(),
thinking_log=thinking_log,
@@ -284,7 +307,7 @@ class TestTuiPresenterState:
assert any(w.startswith("[thinking]") for w in writes), writes
assert any(w == "[render_error] AttributeError" for w in writes), writes
assert not any("rule write failed" in w for w in writes), writes
assert not log.write.called
assert not transcript.mount.called
def test_state_reset_per_worker(self) -> None:
"""state_reset_per_worker [trace]: fresh TuiPresenterState() starts no thinking open."""
@@ -293,7 +316,7 @@ class TestTuiPresenterState:
s1 = TuiPresenterState()
s1.render(
Thinking(sse_id=SID, content="x"),
log=MagicMock(),
transcript=MagicMock(),
tools_log=MagicMock(),
debug_log=MagicMock(),
thinking_log=MagicMock(),
@@ -310,12 +333,12 @@ class TestTuiPresenterState:
"""
from ratatoskr.tui import TuiPresenterState
log = MagicMock()
transcript = MagicMock()
thinking_log = MagicMock()
state = TuiPresenterState()
state.render(
Thinking(sse_id=SID, content="partial"),
log=log,
transcript=transcript,
tools_log=MagicMock(),
debug_log=MagicMock(), thinking_log=thinking_log, raw=False,
)
@@ -323,81 +346,88 @@ class TestTuiPresenterState:
Cancelled(
sse_id=SID, phase="cancelled", turn_id=42, reason="user", partial_message_id=None
),
log=log,
transcript=transcript,
tools_log=MagicMock(),
debug_log=MagicMock(), thinking_log=thinking_log, raw=False,
)
# v0.6.5: streamed thinking + Rule(end) in thinking_log; [cancelled] in transcript.
log_writes = [_text_of(c[0][0]) for c in log.write.call_args_list]
assert any(w.startswith("[cancelled]") for w in log_writes)
transcript_renderables = [_text_of(r) for r in _mounted_renderables(transcript)]
assert any(w.startswith("[cancelled]") for w in transcript_renderables)
# thinking_log got at least Rule(start) + "partial" delta + Rule(end)
assert thinking_log.write.call_count >= 3
def test_done_flushes_tail_and_writes_label(self) -> None:
"""done_flushes_tail_and_writes_label [happy, v0.8.2]:
Text("hi") buffers in text_chunk_buffer (no `\\n`). Done flushes
"hi" tail to transcript, then writes [done] label. v0.8.2 drops
the post-Done Markdown body re-render — streamed text is the
canonical content (no double-print).
def test_text_then_done_mounts_widget_and_finalizes(self) -> None:
"""text_then_done_mounts_widget_and_finalizes [happy, v0.9.0]:
First Text delta mounts a Static(Markdown(buffer)) into the transcript;
Done finalizes the widget reference and mounts a styled [done] label.
No duplicate content (v0.9.0 replaces v0.8.x's flush-on-Done with
live in-place Markdown updates).
"""
from rich.markdown import Markdown
from rich.rule import Rule
from ratatoskr.tui import TuiPresenterState
log = MagicMock()
transcript = MagicMock()
state = TuiPresenterState()
state.render(
Text(sse_id=SID, content="hi"),
log=log,
transcript=transcript,
tools_log=MagicMock(),
debug_log=MagicMock(),
thinking_log=MagicMock(),
raw=False,
)
assert not log.write.called
# v0.9.0: response widget mounted on first Text delta with Markdown wrapper.
assert transcript.mount.called
first_widget = transcript.mount.call_args_list[0][0][0]
assert isinstance(first_widget.content, Markdown)
assert first_widget.content.markup == "hi"
assert state.text_chunk_buffer == "hi"
# Done finalizes: text_chunk_buffer cleared, widget ref released, label mounted.
state.render(
_make_tui_done(),
log=log,
transcript=transcript,
tools_log=MagicMock(),
debug_log=MagicMock(),
thinking_log=MagicMock(),
raw=False,
)
# On Done: tail flush "hi" + [done] label. No Markdown, no Rule.
writes = [c[0][0] for c in log.write.call_args_list]
assert "hi" in writes
writes = _mounted_renderables(transcript)
assert any(_text_of(w).startswith("[done]") for w in writes)
# v0.8.2: no post-Done re-render — no duplicate content.
assert not any(isinstance(w, Markdown) for w in writes)
assert not any(isinstance(w, Rule) for w in writes)
# v0.9.0: response Markdown rendered live during stream — only ONE
# Markdown renderable lands in the transcript (no post-Done re-render).
markdowns = [w for w in writes if isinstance(w, Markdown)]
assert len(markdowns) == 1
assert state.text_chunk_buffer == ""
assert state.current_response_widget is None
def test_raw_flag_skips_markdown(self) -> None:
"""raw_flag_skips_markdown [trace]: raw=True → no Rule, no Markdown."""
"""raw_flag_skips_markdown [v0.9.0]: raw=True → response widget holds
plain str instead of Markdown. Live in-place update still happens;
only the wrapper differs.
"""
from rich.markdown import Markdown
from rich.rule import Rule
from ratatoskr.tui import TuiPresenterState
log = MagicMock()
transcript = MagicMock()
state = TuiPresenterState()
state.render(
Text(sse_id=SID, content="hi"),
log=log,
transcript=transcript,
tools_log=MagicMock(),
debug_log=MagicMock(), thinking_log=MagicMock(), raw=True,
)
state.render(
_make_tui_done(),
log=log,
transcript=transcript,
tools_log=MagicMock(),
debug_log=MagicMock(), thinking_log=MagicMock(), raw=True,
)
writes = [c[0][0] for c in log.write.call_args_list]
assert not any(isinstance(w, Rule) for w in writes)
writes = _mounted_renderables(transcript)
# Raw mode bypasses Markdown entirely — content lives as plain str.
assert not any(isinstance(w, Markdown) for w in writes)
assert "hi" in writes
def test_worker_phase_demoted_to_debug_log(self) -> None:
"""worker_phase_demoted_to_debug_log [trace, v0.5.0]: WorkerPhase → debug_log
@@ -408,17 +438,17 @@ class TestTuiPresenterState:
from ratatoskr.tui import TuiPresenterState
log = MagicMock()
transcript = MagicMock()
debug_log = MagicMock()
state = TuiPresenterState()
state.render(
WorkerPhase(sse_id=SID, phase="streaming", turn_id=42),
log=log,
transcript=transcript,
tools_log=MagicMock(),
debug_log=debug_log, thinking_log=MagicMock(), raw=False,
)
# v0.5.0: WorkerPhase routes to debug_log, NOT transcript.
assert not log.write.called
assert not transcript.mount.called
renderable = debug_log.write.call_args[0][0]
# INV-003: must be a styled Rich Text renderable, not a plain str.
# v0.4.1 retheme: style is now Australis Sea dark-60 ("#86929d") instead
@@ -443,12 +473,12 @@ class TestTuiPresenterState:
"""
from ratatoskr.tui import TuiPresenterState
log = MagicMock()
transcript = MagicMock()
tools_log = MagicMock()
state = TuiPresenterState()
state.render(
ToolStart(sse_id=SID, name="read_file", arguments={"path": "/x"}),
log=log,
transcript=transcript,
tools_log=tools_log,
debug_log=MagicMock(), thinking_log=MagicMock(), raw=False,
)
@@ -456,85 +486,95 @@ class TestTuiPresenterState:
assert tools_log.write.called
assert _text_of(tools_log.write.call_args[0][0]).startswith("· tool_start:")
# INV-014: transcript was NOT written to
assert not log.write.called
assert not transcript.mount.called
def test_tool_result_routes_to_tools_log(self) -> None:
"""tool_result_routes_to_tools_log [INV-014]: ToolResult → tools_log, NOT transcript."""
from ratatoskr.tui import TuiPresenterState
log = MagicMock()
transcript = MagicMock()
tools_log = MagicMock()
state = TuiPresenterState()
state.render(
ToolResult(sse_id=SID, name="read_file", result="ok", duration_ms=12),
log=log,
transcript=transcript,
tools_log=tools_log,
debug_log=MagicMock(), thinking_log=MagicMock(), raw=False,
)
assert tools_log.write.called
assert _text_of(tools_log.write.call_args[0][0]).startswith("· tool_result:")
assert not log.write.called
assert not transcript.mount.called
def test_text_event_buffers_until_newline(self) -> None:
"""text_event_buffers_until_newline [v0.8.1]: Text deltas without
`\\n` accumulate in text_chunk_buffer; no log write yet.
def test_text_first_delta_mounts_response_widget(self) -> None:
"""text_first_delta_mounts_response_widget [v0.9.0]: first Text delta
mounts a Static carrying Markdown(buffer) into the transcript. The
text_chunk_buffer holds the accumulated content for the next delta's
in-place update.
"""
from rich.markdown import Markdown
from ratatoskr.tui import TuiPresenterState
log = MagicMock()
transcript = MagicMock()
tools_log = MagicMock()
state = TuiPresenterState()
state.render(
Text(sse_id=SID, content="hello"),
log=log,
transcript=transcript,
tools_log=tools_log,
debug_log=MagicMock(),
thinking_log=MagicMock(),
raw=False,
)
# v0.8.1: buffered, not written until `\n` or Done.
assert state.text_chunk_buffer == "hello"
assert not log.write.called
assert transcript.mount.call_count == 1
widget = transcript.mount.call_args[0][0]
assert isinstance(widget.content, Markdown)
assert widget.content.markup == "hello"
assert state.current_response_widget is widget
assert not tools_log.write.called
def test_text_flushes_on_newline(self) -> None:
"""text_flushes_on_newline [v0.8.1]: a delta carrying `\\n` flushes
the accumulated buffer as ONE line to log (transcript).
def test_text_subsequent_deltas_update_in_place(self) -> None:
"""text_subsequent_deltas_update_in_place [v0.9.0]: deltas after the
first do NOT mount a new widget — they update the existing widget's
Markdown content in place. The text_chunk_buffer accumulates.
"""
from ratatoskr.tui import TuiPresenterState
log = MagicMock()
transcript = MagicMock()
state = TuiPresenterState()
for tok in ("Hel", "lo", " ", "world", "\n"):
for tok in ("Hel", "lo", " ", "world"):
state.render(
Text(sse_id=SID, content=tok),
log=log,
transcript=transcript,
tools_log=MagicMock(),
debug_log=MagicMock(),
thinking_log=MagicMock(),
raw=False,
)
writes = [c[0][0] for c in log.write.call_args_list]
# "Hello world" coalesces to ONE log entry.
assert writes == ["Hello world"]
assert state.text_chunk_buffer == ""
# Exactly ONE mount (the first delta); subsequent deltas update.
assert transcript.mount.call_count == 1
assert state.text_chunk_buffer == "Hello world"
# Widget reference held; buffer is the source of truth re-rendered
# into Markdown(...) for each Static.update call.
assert state.current_response_widget is not None
def test_duration_format_seconds(self) -> None:
"""duration_format_seconds [trace]: Done(duration_ms=5467) → label has "duration=5.5s"."""
from ratatoskr.tui import TuiPresenterState
log = MagicMock()
transcript = MagicMock()
state = TuiPresenterState()
state.render(
_make_tui_done(duration_ms=5467),
log=log,
transcript=transcript,
tools_log=MagicMock(),
debug_log=MagicMock(), thinking_log=MagicMock(), raw=True,
)
done_line = next(
_text_of(c[0][0])
for c in log.write.call_args_list
if _text_of(c[0][0]).startswith("[done]")
_text_of(r)
for r in _mounted_renderables(transcript)
if _text_of(r).startswith("[done]")
)
assert "duration=5.5s" in done_line
assert "duration_ms=5467" not in done_line
@@ -543,7 +583,7 @@ class TestTuiPresenterState:
"""usage_format_unicode_arrow [trace]: TUI Done label uses → (Unicode), not -> (ASCII)."""
from ratatoskr.tui import TuiPresenterState
log = MagicMock()
transcript = MagicMock()
state = TuiPresenterState()
usage = {
"prompt_tokens": 6756,
@@ -553,32 +593,172 @@ class TestTuiPresenterState:
}
state.render(
_make_tui_done(usage=usage),
log=log,
transcript=transcript,
tools_log=MagicMock(),
debug_log=MagicMock(), thinking_log=MagicMock(), raw=True,
)
done_line = next(
_text_of(c[0][0])
for c in log.write.call_args_list
if _text_of(c[0][0]).startswith("[done]")
_text_of(r)
for r in _mounted_renderables(transcript)
if _text_of(r).startswith("[done]")
)
assert "usage 6756 in → 126 out (6882 total, 0 cached)" in done_line
class TestPresenterAuditLogging:
"""v0.10.0 — per-event audit lines + turn-summary in the debug pane.
The presenter emits one debug-pane line per arriving event (Text and
Thinking are aggregated into the turn-summary instead of per-delta to
avoid drowning the pane at token rate).
"""
def test_worker_phase_emits_audit_line(self) -> None:
"""worker_phase_emits_audit_line: WorkerPhase arrival adds an audit
line to debug_log alongside the existing `· worker_phase:` entry.
Audit line shape: `[HH:MM:SS.fff] workerphase sse_id=N:M …`.
"""
from ratatoskr.tui import TuiPresenterState
debug_log = MagicMock()
state = TuiPresenterState()
state.render(
WorkerPhase(sse_id=SID, phase="streaming", turn_id=42),
transcript=MagicMock(),
tools_log=MagicMock(),
debug_log=debug_log,
thinking_log=MagicMock(),
raw=False,
)
# Two writes: audit line + worker_phase telemetry.
assert debug_log.write.call_count == 2
audit_line = _text_of(debug_log.write.call_args_list[0][0][0])
assert "workerphase" in audit_line
assert "sse_id=42:5" in audit_line
assert "phase=streaming" in audit_line
def test_tool_start_emits_audit_line(self) -> None:
"""tool_start_emits_audit_line: ToolStart adds one audit line to
debug_log even though the tool event itself routes to tools_log.
"""
from ratatoskr.tui import TuiPresenterState
debug_log = MagicMock()
state = TuiPresenterState()
state.render(
ToolStart(sse_id=SID, name="read_file", arguments={"path": "/x"}),
transcript=MagicMock(),
tools_log=MagicMock(),
debug_log=debug_log,
thinking_log=MagicMock(),
raw=False,
)
assert debug_log.write.call_count == 1
audit_line = _text_of(debug_log.write.call_args[0][0])
assert "toolstart" in audit_line
assert "sse_id=42:5" in audit_line
assert "name=read_file" in audit_line
def test_text_delta_counted_not_per_event_audit_line(self) -> None:
"""text_delta_counted_not_per_event_audit_line: a Text delta does
NOT emit a per-event audit line (token-rate would drown the pane);
instead it bumps text_delta_count / text_byte_count for the turn-
summary at Done.
"""
from ratatoskr.tui import TuiPresenterState
debug_log = MagicMock()
state = TuiPresenterState()
state.render(
Text(sse_id=SID, content="hello world"),
transcript=MagicMock(),
tools_log=MagicMock(),
debug_log=debug_log,
thinking_log=MagicMock(),
raw=False,
)
# No debug-pane writes — text deltas are silent at token rate.
assert not debug_log.write.called
assert state.text_delta_count == 1
assert state.text_byte_count == len("hello world")
def test_done_emits_turn_summary_line(self) -> None:
"""done_emits_turn_summary_line: when Done arrives the presenter
emits a `turn_summary` line aggregating per-delta Text + Thinking
counters. The shape exposes the totals that per-event audit lines
elided.
"""
from ratatoskr.tui import TuiPresenterState
debug_log = MagicMock()
state = TuiPresenterState()
# 3 Text deltas + 2 Thinking deltas, then Done.
state.render(
Text(sse_id=SID, content="a"),
transcript=MagicMock(), tools_log=MagicMock(),
debug_log=debug_log, thinking_log=MagicMock(), raw=False,
)
state.render(
Text(sse_id=SID, content="bc"),
transcript=MagicMock(), tools_log=MagicMock(),
debug_log=debug_log, thinking_log=MagicMock(), raw=False,
)
state.render(
Thinking(sse_id=SID, content="thought\n"),
transcript=MagicMock(), tools_log=MagicMock(),
debug_log=debug_log, thinking_log=MagicMock(), raw=False,
)
state.render(
_make_tui_done(),
transcript=MagicMock(), tools_log=MagicMock(),
debug_log=debug_log, thinking_log=MagicMock(), raw=False,
)
writes = [_text_of(c[0][0]) for c in debug_log.write.call_args_list]
summary = next(w for w in writes if "turn_summary" in w)
assert "text_deltas=2" in summary
assert "text_bytes=3" in summary # "a" + "bc"
assert "thinking_deltas=1" in summary
assert "elapsed_ms=" in summary
def _text_of(write_arg: object) -> str:
"""Extract plain text from a RichLog.write() arg (str or rich.text.Text).
Issue #12 wraps demoted-telemetry entries in `rich.text.Text(..., style="dim")`
so the RichLog can apply dim styling; non-demoted writes stay as plain str.
Tests that want to assert against content need both shapes flattened.
v0.9.0: also extracts plain text from Markdown wrappers (the streaming-text
response path uses Markdown(buffer) now; tests assert against the source
markup, which lives in `Markdown.markup`).
"""
from rich.markdown import Markdown
from rich.text import Text as RichText
if isinstance(write_arg, RichText):
return write_arg.plain
if isinstance(write_arg, Markdown):
return write_arg.markup
if isinstance(write_arg, str):
return write_arg
return "" # Markdown / Rule / etc. — not text content
return "" # Rule / etc. — not text content
def _mounted_renderables(transcript_mock: MagicMock) -> list:
"""v0.9.0: TuiPresenterState now mounts Static widgets into the transcript
VerticalScroll instead of writing renderables to a RichLog. Tests using a
MagicMock transcript inspect `transcript.mount.call_args_list`; each call's
first positional arg is the Static child whose `.content` carries the
Markdown / RichText / str that pre-v0.9.0 would have been the write arg.
Returns those renderables in mount-call order so tests can assert on them
with the same shape they used for `log.write.call_args_list` previously.
"""
out: list = []
for call in transcript_mock.mount.call_args_list:
for child in call.args:
renderable = getattr(child, "content", child)
out.append(renderable)
return out
def _make_tui_done(*, duration_ms: int = 1, usage: dict[str, int] | None = None) -> Done:
@@ -602,15 +782,15 @@ def _make_tui_done(*, duration_ms: int = 1, usage: dict[str, int] | None = None)
class TestCancelViaSse:
@respx.mock
async def test_happy_cancel(self) -> None:
"""happy_cancel [happy,tracer]: 200 OK → returns None; log has no [cancel_failed]."""
"""happy_cancel [happy,tracer]: 200 OK → returns None; transcript has no [cancel_failed]."""
respx.post("https://w.example/sessions/s-1/turns/42/cancel").mock(
return_value=httpx.Response(200, json=_CANCEL_OK_RESP)
)
log = MagicMock()
transcript = MagicMock()
async with httpx.AsyncClient(base_url="https://w.example") as client:
result = await _cancel_via_sse(client, "s-1", 42, log=log)
result = await _cancel_via_sse(client, "s-1", 42, transcript=transcript)
assert result is None
log.write.assert_not_called()
transcript.mount.assert_not_called()
@respx.mock
async def test_cancel_failed_500(self) -> None:
@@ -618,10 +798,10 @@ class TestCancelViaSse:
respx.post("https://w.example/sessions/s-1/turns/42/cancel").mock(
return_value=httpx.Response(500, content=b"boom")
)
log = MagicMock()
transcript = MagicMock()
async with httpx.AsyncClient(base_url="https://w.example") as client:
await _cancel_via_sse(client, "s-1", 42, log=log)
line = log.write.call_args[0][0]
await _cancel_via_sse(client, "s-1", 42, transcript=transcript)
line = transcript.mount.call_args[0][0].content
assert "[cancel_failed]" in line
assert "CancelFailed" in line
@@ -631,10 +811,10 @@ class TestCancelViaSse:
respx.post("https://w.example/sessions/s-1/turns/42/cancel").mock(
return_value=httpx.Response(409)
)
log = MagicMock()
transcript = MagicMock()
async with httpx.AsyncClient(base_url="https://w.example") as client:
await _cancel_via_sse(client, "s-1", 42, log=log)
line = log.write.call_args[0][0]
await _cancel_via_sse(client, "s-1", 42, transcript=transcript)
line = transcript.mount.call_args[0][0].content
assert "[cancel_failed]" in line
assert "CancelAlreadyCompleted" in line
@@ -644,13 +824,54 @@ class TestCancelViaSse:
respx.post("https://w.example/sessions/s-1/turns/42/cancel").mock(
side_effect=httpx.ConnectError("network down")
)
log = MagicMock()
transcript = MagicMock()
async with httpx.AsyncClient(base_url="https://w.example") as client:
await _cancel_via_sse(client, "s-1", 42, log=log)
line = log.write.call_args[0][0]
await _cancel_via_sse(client, "s-1", 42, transcript=transcript)
line = transcript.mount.call_args[0][0].content
assert "[cancel_failed]" in line
assert "ConnectError" in line
@respx.mock
async def test_audit_callback_records_lifecycle(self) -> None:
"""audit_callback_records_lifecycle [v0.10.0]: when the caller passes
an `audit` callback, _cancel_via_sse emits two lines on the happy
path (`cancel_post issued …` + `cancel_post ok …`) and two lines on
the failure path (`issued` + `failed …`). Gives the debug pane a
complete cancel-POST timeline.
"""
respx.post("https://w.example/sessions/s-1/turns/42/cancel").mock(
return_value=httpx.Response(200, json=_CANCEL_OK_RESP)
)
transcript = MagicMock()
audit_lines: list = []
async with httpx.AsyncClient(base_url="https://w.example") as client:
await _cancel_via_sse(
client, "s-1", 42, transcript=transcript, audit=audit_lines.append
)
assert len(audit_lines) == 2
assert audit_lines[0].startswith("cancel_post issued ")
assert "session_id=s-1" in audit_lines[0]
assert "turn_id=42" in audit_lines[0]
assert audit_lines[1] == "cancel_post ok turn_id=42"
@respx.mock
async def test_audit_callback_records_failure(self) -> None:
"""audit_callback_records_failure [v0.10.0]: failure path emits
`cancel_post issued` then `cancel_post failed …` with exception type.
"""
respx.post("https://w.example/sessions/s-1/turns/42/cancel").mock(
return_value=httpx.Response(500, content=b"boom")
)
transcript = MagicMock()
audit_lines: list = []
async with httpx.AsyncClient(base_url="https://w.example") as client:
await _cancel_via_sse(
client, "s-1", 42, transcript=transcript, audit=audit_lines.append
)
assert len(audit_lines) == 2
assert audit_lines[0].startswith("cancel_post issued ")
assert audit_lines[1].startswith("cancel_post failed turn_id=42 CancelFailed")
class TestAppMount:
"""on_mount narrows per issue #6: only identity-widget population.
@@ -714,18 +935,18 @@ class TestLayoutShape:
assert row is not None
async def test_left_column_content_only(self) -> None:
"""left_column_content_only [v0.6.5]: left column = transcript + prompt
+ current-text (streaming text Static). thinking-current Static
removed entirely as of v0.6.5.
"""left_column_content_only [v0.9.0]: left column = transcript-scroll
VerticalScroll + prompt Input. thinking-current Static removed in
v0.6.5; transcript RichLog replaced by VerticalScroll in v0.9.0.
"""
from textual.containers import Vertical
from textual.widgets import Input, RichLog
from textual.containers import Vertical, VerticalScroll
from textual.widgets import Input
app = _resolved_app(_args_new(), session_id="s-new12345", agent_id="mimir")
async with app.run_test() as pilot:
await pilot.pause()
left = app.query_one("#left-column", Vertical)
transcript = app.query_one("#transcript", RichLog)
transcript = app.query_one("#transcript-scroll", VerticalScroll)
prompt = app.query_one("#prompt", Input)
assert transcript in left.walk_children()
assert prompt in left.walk_children()
@@ -750,7 +971,7 @@ class TestLayoutShape:
assert tools_tab is not None
async def test_tools_log_inside_tools_tab(self) -> None:
"""tools_log_inside_tools_tab: tools-log RichLog is a descendant of tools-tab TabPane."""
"""tools_log_inside_tools_tab: tools-transcript RichLog is a descendant of tools-tab TabPane."""
from textual.widgets import RichLog, TabPane
app = _resolved_app(_args_new(), session_id="s-new12345", agent_id="mimir")
@@ -801,7 +1022,7 @@ class TestLayoutShape:
)
async def test_debug_tab_exists(self) -> None:
"""debug_tab_exists [v0.5.0]: right column has Debug TabPane + #debug-log RichLog."""
"""debug_tab_exists [v0.5.0]: right column has Debug TabPane + #debug-transcript RichLog."""
from textual.widgets import RichLog, TabPane
app = _resolved_app(_args_new(), session_id="s-new12345", agent_id="mimir")
@@ -823,36 +1044,45 @@ class TestLayoutShape:
assert app.query_one("#side-panes", TabbedContent).active == "debug-tab"
async def test_done_label_styled_success(self) -> None:
"""done_label_styled_success [v0.5.1]: [done] label renders in Aurora green."""
"""done_label_styled_success [v0.9.0]: [done] label mounts as Static
carrying a RichText with Aurora green style. Inspect the mounted
Static's `.content`.
"""
from rich.text import Text as RichText
from textual.containers import VerticalScroll
from textual.widgets import RichLog
app = _resolved_app(_args_new(), session_id="s-new12345", agent_id="mimir")
async with app.run_test() as pilot:
await pilot.pause()
# Probe the presenter directly — write a Done via state.render.
from ratatoskr.tui import TuiPresenterState
log = app.query_one("#transcript", RichLog)
transcript = app.query_one("#transcript-scroll", VerticalScroll)
state = TuiPresenterState()
seen: list = []
orig = log.write
log.write = lambda c, *a, **kw: (seen.append(c), orig(c, *a, **kw))[1]
mounted: list = []
orig_mount = transcript.mount
def spy_mount(*ch, **kw):
mounted.extend(ch)
return orig_mount(*ch, **kw)
transcript.mount = spy_mount # type: ignore[method-assign]
state.render(
_make_tui_done(),
log=log,
transcript=transcript,
tools_log=app.query_one("#tools-log", RichLog),
debug_log=app.query_one("#debug-log", RichLog),
thinking_log=MagicMock(),
raw=True,
)
done = next(
c for c in seen
if isinstance(c, RichText) and _text_of(c).startswith("[done]")
w.content for w in mounted
if isinstance(getattr(w, "content", None), RichText)
and _text_of(w.content).startswith("[done]")
)
assert done.style == "#16B866" # Aurora green
async def test_empty_state_placeholders_present(self) -> None:
"""empty_state_placeholders_present [v0.5.1]: tools-log + debug-log show
"""empty_state_placeholders_present [v0.5.1]: tools-transcript + debug-transcript show
placeholder lines before any turn fires."""
from textual.widgets import RichLog
@@ -1061,11 +1291,13 @@ async def _submit_and_wait(app: RatatoskrApp, pilot, content: str) -> None:
class TestStreamTurnWorker:
@respx.mock
async def test_happy_text_done_no_double_print(self, monkeypatch: pytest.MonkeyPatch) -> None:
"""happy_text_done_no_double_print [happy,tracer, v0.8.2]:
Text("hello") buffers; on Done, "hello" flushes as tail to transcript
+ [done] label. v0.8.2 drops the post-Done Markdown body re-render
(was double-printing the response — streamed text + Markdown twice).
Only the turn-header Rule remains in the transcript.
"""happy_text_done_no_double_print [happy,tracer, v0.9.0]:
Text("hello") mounts a Static(Markdown("hello")) into the transcript;
Done mounts a [done] label Static. The Markdown is rendered live (one
widget for the whole stream, updated in place), so there is NO
post-Done re-render — exactly ONE Markdown renderable lands in the
transcript for the response body. v0.9.0 supersedes v0.8.2's
drop-Markdown patch with proper live rendering.
"""
stream = _sse_chunk("42:1", {"type": "text", "content": "hello"}) + _sse_chunk(
"42:2", _DONE_BODY
@@ -1083,23 +1315,25 @@ class TestStreamTurnWorker:
assert app.state == "idle"
from rich.markdown import Markdown
# "hello" appears as a tail-flush; [done] label fires.
assert any(w == "hello" for w in writes)
assert any("[done]" in str(w) for w in writes)
# v0.8.2: NO Markdown body re-render (was the duplicate).
assert not any(isinstance(w, Markdown) for w in writes)
# The turn-header Rule is written to all 4 panes; we still expect
# SOME Rules in the spy (one per pane), but NOT the post-Done
# separator Rule that pre-v0.8.2 wrote.
# We rely on _spy_writes counting turn-header Rules only.
# The response body lives as ONE Markdown renderable mounted into
# the transcript; live updates happen via Static.update, not via
# re-mount, so there's exactly one Markdown in the spy stream.
markdowns = [w for w in writes if isinstance(w, Markdown)]
assert len(markdowns) == 1, (
f"v0.9.0: expected exactly ONE Markdown mounted, got {len(markdowns)}"
)
assert markdowns[0].markup == "hello"
# [done] label fires too.
assert any("[done]" in _text_of(w) for w in writes)
@respx.mock
async def test_raw_flag_skips_markdown_render(self, monkeypatch: pytest.MonkeyPatch) -> None:
"""raw_flag_skips_markdown_render [trace, v0.6.0]:
With --raw, no Markdown render. A turn-header Rule IS still written
(v0.6.0 INV — turn correlation lives in every pane). The post-Done
Rule(separator) is suppressed; accumulated streamed text is written
as a plain string instead.
"""raw_flag_skips_markdown_render [trace, v0.9.0]:
With --raw, the response widget holds plain str instead of Markdown.
Turn-header markers still appear in every pane: 3 RichLog panes
receive a Rule, the transcript-scroll receives a Static-wrapped
RichText (mounted, not written), giving 3 Rules in the captured
writes list.
"""
stream = _sse_chunk("42:1", {"type": "text", "content": "hi"}) + _sse_chunk(
"42:2", _DONE_BODY
@@ -1117,11 +1351,11 @@ class TestStreamTurnWorker:
# No Markdown in raw mode.
assert not any(isinstance(w, Markdown) for w in writes)
# Only turn-header Rules — one per pane (transcript + tools +
# debug + thinking = 4). No post-Done separator Rule.
# 3 Rules — one per RichLog pane (tools / debug / thinking).
# Transcript-scroll uses a Static turn-header Markdown alternative.
rules = [w for w in writes if isinstance(w, Rule)]
assert len(rules) == 4, f"expected 4 turn-header Rules, got {len(rules)}"
# Accumulated text "hi" written as plain string post-Done.
assert len(rules) == 3, f"expected 3 turn-header Rules, got {len(rules)}"
# Accumulated text "hi" mounted as plain str into transcript.
assert "hi" in writes
@respx.mock
@@ -1495,8 +1729,14 @@ class TestActionInterrupt:
await pilot.pause(0.02)
# Give _cancel_via_sse time to write the [cancel_failed] line
await pilot.pause(0.05)
log = app.query_one("#transcript", RichLog)
rendered = "\n".join(str(strip.text) for strip in log.lines)
from textual.containers import VerticalScroll
from textual.widgets import Static
transcript = app.query_one("#transcript-scroll", VerticalScroll)
rendered = "\n".join(
str(child.content)
for child in transcript.children
if isinstance(child, Static)
)
assert "[cancel_failed]" in rendered
assert app.state == "cancelling"
stream_gate.set() # let stream finish for teardown
Generated
+1 -1
View File
@@ -968,7 +968,7 @@ wheels = [
[[package]]
name = "ratatoskr"
version = "0.8.2"
version = "0.10.0"
source = { editable = "." }
dependencies = [
{ name = "httpx" },