feat(R04): run every Cowork turn through ConversationApplicationService

R04-T03 — the turn lifecycle, extracted from `core/chat_agent.py::run_cowork`
into `application/conversations/`. The 260-line body mixed the lifecycle (step
budget, cancel checks, guard -> preview -> gate -> execute ordering, sandbox
tidy-up) with the machinery doing each step, and reaching any of it meant
standing up a Qt widget and a worker thread. It is now a plain object driven
through two Protocols and six callables (`turn_runtime.py`), with the concrete
`core/*` wiring confined to `core_runtime_adapter.py` — the same shape R03 used
for routing. Faithful port, not an improvement pass: where the original had a
quirk (the step-ceiling note only merges into the answer when the last message
is the assistant's) the quirk is preserved and commented.

R04-T04 — `ui/cowork_tab.py::build_job` no longer calls run_cowork. It captures
the widget's state at submit time, builds the request via the new
`cowork_turn_request.py` and executes it. `execute(..., messages=...)` hands the
widget's own list over because `_reattach_running_turn` replays from it WHILE
the worker appends and `_finalize_turn` slices it afterwards — a private list
would break both silently.

R04-T05 — `core/task_executors.py`'s cowork branch shares the same engine. All
five unattended-run behaviours stay put (plan reminder, history_ready, History
autosave per assistant message, timeout notice, plan_incomplete_reason), and
`_unattended_prompt` now expresses the load-bearing prefix order in one
readable call instead of three successive rebindings.

Verification: 74 new tests (364 passed, 1 skipped overall; check_imports PASS).
The two that matter most:
- `test_conversation_service_parity.py` runs the same scripted turn through
  run_cowork AND the service and compares the event stream, the resulting
  conversation and the advertised tool list across 7 scenarios;
- `test_task_executor_turn.py` was written BEFORE the migration and passed 8/8
  against the old code, then unchanged against the new.

Known: `ui/cowork_tab.py` (416 -> 455) and `core/task_executors.py` (476 -> 524)
stay above the 400-LOC limit. Both were already over it before this change;
bringing them under needs the R08 / R07 decompositions.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
2026-08-23 13:14:15 +09:00
co-authored by Claude Opus 5
parent 19e6b4deb2
commit 3665135c38
16 changed files with 2452 additions and 48 deletions
+141
View File
@@ -0,0 +1,141 @@
"""Offline test doubles for the R04 turn runtime seams.
Sits beside ``fake_provider.py``/``fake_tool_executor.py`` (R01-T02) and plays
the same role one level up: those fake a *provider*, these fake the ports
``ConversationApplicationService`` is driven through
(``application/conversations/turn_runtime.py``).
Deliberately dumb — they record what they were asked and return canned answers.
A failing test then points at the service under test rather than at a mock
framework's configuration.
"""
from __future__ import annotations
from typing import Any, Dict, List, Optional, Tuple
from cowork_local.domain.agents.agent_event import ToolPreview
from cowork_local.domain.agents.conversation_execution_request import (
ConversationExecutionRequest,
)
class FakeSpec:
"""An advertised tool. The service only ever reads ``.name`` off a spec."""
def __init__(self, name: str) -> None:
self.name = name
class FakeReply:
"""One programmed provider answer."""
def __init__(self, content: str = "", tool_calls=None, chunks=None, reasoning: str = ""):
self.content = content
self.tool_calls = tool_calls or []
# Default to streaming the whole content as a single chunk, which is what
# a non-streaming gateway effectively does.
self.chunks = chunks if chunks is not None else ([content] if content else [])
self.reasoning = reasoning
class FakeModelCall:
""":class:`ModelCallPort` returning programmed replies in order.
A programmed entry may be an exception instead of a reply, which is how a
test simulates the gateway dying mid-turn.
"""
def __init__(self, replies: List[Any]) -> None:
self.replies = list(replies)
self.calls: List[Dict[str, Any]] = []
def call(self, messages, tools, on_text=None, on_reasoning=None, cancel=None):
# Snapshot the messages: the service keeps mutating its own list, so
# storing it by reference would make every recorded call look identical.
self.calls.append({"messages": [dict(m) for m in messages],
"tool_names": [getattr(t, "name", "") for t in tools]})
reply = self.replies.pop(0) if self.replies else FakeReply(content="(default)")
if isinstance(reply, BaseException):
raise reply
if reply.reasoning and on_reasoning:
on_reasoning(reply.reasoning)
for chunk in reply.chunks:
if on_text and chunk:
on_text(chunk)
assistant: Dict[str, Any] = {"role": "assistant", "content": reply.content}
if reply.tool_calls:
assistant["tool_calls"] = reply.tool_calls
return assistant
class FakeToolRuntime:
""":class:`ToolRuntimePort` over an imaginary output folder."""
def __init__(self, specs=("save_file", "run_command", "update_plan"),
results: Optional[Dict[str, Dict[str, Any]]] = None,
removed: Tuple[str, ...] = (), added: Tuple[str, ...] = ()) -> None:
self._specs = [FakeSpec(n) for n in specs]
self._results = results or {}
self._removed, self._added = removed, added
self.executed: List[Tuple[str, Dict[str, Any]]] = []
self.finalize_calls: List[Dict[str, Any]] = []
# When set, every executed tool streams this string through ``on_output``.
self.emit_output: Optional[str] = None
def specs(self, allowed_tools=None):
if allowed_tools is None:
return list(self._specs)
return [s for s in self._specs if s.name in allowed_tools]
def preview(self, name, args):
return ToolPreview(kind="info", title=name, text=str(args))
def execute(self, name, args, on_output=None, cancel=None):
self.executed.append((name, dict(args)))
if self.emit_output and on_output:
on_output(self.emit_output)
return dict(self._results.get(name, {"ok": True, "output": f"{name} ok"}))
def snapshot(self):
return "before"
def finalize(self, before, cancelled=False):
self.finalize_calls.append({"before": before, "cancelled": cancelled})
return list(self._removed), list(self._added)
# --------------------------------------------------------------------------- #
# Small helpers shared by the turn tests.
# --------------------------------------------------------------------------- #
def make_request(**overrides) -> ConversationExecutionRequest:
"""A minimal valid request; each test overrides only what it exercises."""
base: Dict[str, Any] = {"turn_id": "t1", "session_id": "s1", "prompt": "do it"}
base.update(overrides)
return ConversationExecutionRequest(**base)
def run_turn(service, request=None, cancel=None):
"""Execute a turn and return ``(result, events)``."""
events: List[Any] = []
result = service.execute(request or make_request(), events.append, cancel=cancel)
return result, events
def events_of_type(events, cls):
"""Every emitted event of one type, in order."""
return [e for e in events if isinstance(e, cls)]
def tool_turn(tool_name: str = "save_file", args=None, **tool_kwargs):
"""A turn that calls one tool and then answers — ``(model, tools)``."""
calls = [{"id": "c1", "name": tool_name, "arguments": args or {"filename": "a.md"}}]
model = FakeModelCall([FakeReply(content="working", tool_calls=calls),
FakeReply(content="done")])
return model, FakeToolRuntime(**tool_kwargs)
__all__ = [
"FakeSpec", "FakeReply", "FakeModelCall", "FakeToolRuntime",
"make_request", "run_turn", "events_of_type", "tool_turn",
]
@@ -0,0 +1,207 @@
"""R04-T03 (c) — the service must behave exactly like ``run_cowork``.
The unit tests prove the loop follows the rules I wrote down. They cannot prove
those rules are the ones the shipped runtime actually follows. This file does:
each test scripts one provider, runs the SAME turn twice — once through
``core/chat_agent.py::run_cowork``, once through
``ConversationApplicationService`` wired by ``core_runtime_adapter`` — and
compares the emitted event stream, the resulting conversation and the tool list
the model was shown.
Anything the port got wrong (a missing event, a reordered guard, a different
tool set, a changed message) fails here rather than in front of a user. The only
allowed difference is the extra ``turn_completed`` event R04 introduces, which
has no legacy consumer.
"""
from __future__ import annotations
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple
from cowork_local.application.conversations.core_runtime_adapter import (
build_cowork_conversation_service,
legacy_event_sink,
)
from cowork_local.core import chat_agent
from cowork_local.domain.agents.conversation_execution_request import (
ConversationExecutionRequest,
)
from cowork_local.tests.fakes.fake_provider import FakeProvider
_USER_TURN = [{"role": "user", "content": "make me a report"}]
class _FakeGate:
"""Stands in for ``core/permissions.py::PermissionGate``."""
def __init__(self, approve: bool) -> None:
self.approve = approve
self.requests: List[Dict[str, Any]] = []
def request(self, action: Dict[str, Any]) -> bool:
self.requests.append(action)
return self.approve
def _normalise(events: List[Dict[str, Any]], out_dir: Path) -> List[Dict[str, Any]]:
"""Replace the run's own output path with a placeholder.
The two runs write into different temp folders, so absolute paths in
``tool_result``/``outputs_*`` events differ by construction. Everything else
must match verbatim.
"""
marker, raw = "<OUT>", str(out_dir)
def scrub(value: Any) -> Any:
if isinstance(value, str):
return value.replace(raw, marker).replace(raw.replace("\\", "/"), marker)
if isinstance(value, list):
return [scrub(v) for v in value]
if isinstance(value, dict):
return {k: scrub(v) for k, v in value.items()}
return value
return [scrub(e) for e in events]
def _run_legacy(tmp_path: Path, provider: FakeProvider, *, allowed_tools=None,
gate: Optional[_FakeGate] = None, max_steps: int = 30
) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]], List[str]]:
"""Run the turn through the existing ``run_cowork``."""
out_dir = tmp_path / "legacy"
out_dir.mkdir(parents=True, exist_ok=True)
events: List[Dict[str, Any]] = []
messages = [dict(m) for m in _USER_TURN]
chat_agent.run_cowork(
provider, messages, out_dir, events.append, title="Report",
security_config=None, allowed_tools=allowed_tools, gate=gate, max_steps=max_steps,
)
tool_names = [t.name for t in (provider.last_tools or [])]
return _normalise(events, out_dir), messages, tool_names
def _run_service(tmp_path: Path, provider: FakeProvider, *, allowed_tools=None,
gate: Optional[_FakeGate] = None, max_steps: int = 30
) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]], List[str]]:
"""Run the same turn through the application service."""
out_dir = tmp_path / "service"
out_dir.mkdir(parents=True, exist_ok=True)
events: List[Dict[str, Any]] = []
service = build_cowork_conversation_service(
provider, out_dir, events.append, title="Report", security_config=None, gate=gate)
request = ConversationExecutionRequest(
turn_id="t1", session_id="s1",
# run_cowork receives the user message already appended; the request
# carries the history and this turn's prompt separately.
messages=_USER_TURN[:-1], prompt=_USER_TURN[-1]["content"],
output_dir=out_dir, allowed_tools=allowed_tools, max_steps=max_steps,
gate_mode="confirm" if gate is not None else "auto",
)
result = service.execute(request, legacy_event_sink(events.append))
# The end-of-turn event is new in R04 and has no legacy counterpart.
kept = [e for e in events if e.get("type") != "turn_completed"]
tool_names = [t.name for t in (provider.last_tools or [])]
return _normalise(kept, out_dir), list(result.messages), tool_names
def _assert_parity(tmp_path: Path, script, *, approve: Optional[bool] = None, **kwargs) -> None:
"""Script two identical providers, run both paths, compare everything."""
legacy_provider, service_provider = FakeProvider(), FakeProvider()
script(legacy_provider)
script(service_provider)
legacy_gate = _FakeGate(approve) if approve is not None else None
service_gate = _FakeGate(approve) if approve is not None else None
legacy_events, legacy_messages, legacy_tools = _run_legacy(
tmp_path, legacy_provider, gate=legacy_gate, **kwargs)
service_events, service_messages, service_tools = _run_service(
tmp_path, service_provider, gate=service_gate, **kwargs)
assert service_events == legacy_events
assert service_messages == legacy_messages
assert service_tools == legacy_tools
if legacy_gate is not None and service_gate is not None:
assert [r["name"] for r in service_gate.requests] == \
[r["name"] for r in legacy_gate.requests]
# --------------------------------------------------------------------------- #
# Scenarios.
# --------------------------------------------------------------------------- #
def test_a_plain_answer_turn_behaves_identically(tmp_path: Path) -> None:
def script(provider: FakeProvider) -> None:
provider.queue_response(content="Here you go.", chunks=["Here ", "you go."])
_assert_parity(tmp_path, script)
def test_a_save_file_turn_behaves_identically(tmp_path: Path) -> None:
def script(provider: FakeProvider) -> None:
provider.queue_response(
content="Writing it.",
tool_calls=[{"id": "c1", "name": "save_file",
"arguments": {"filename": "report.md", "content": "# Report\n"}}],
)
provider.queue_response(content="Saved.")
_assert_parity(tmp_path, script)
def test_an_update_plan_turn_behaves_identically(tmp_path: Path) -> None:
def script(provider: FakeProvider) -> None:
provider.queue_response(
content="Planning.",
tool_calls=[{"id": "c1", "name": "update_plan",
"arguments": {"steps": [{"title": "Draft", "status": "running"},
{"title": "Ship", "status": "pending"}]}}],
)
provider.queue_response(content="Done.")
_assert_parity(tmp_path, script)
def test_a_reasoning_only_reply_behaves_identically(tmp_path: Path) -> None:
def script(provider: FakeProvider) -> None:
provider.queue_response(content="", reasoning="thinking hard")
_assert_parity(tmp_path, script)
def test_restricting_the_tool_scope_advertises_the_same_tools(tmp_path: Path) -> None:
def script(provider: FakeProvider) -> None:
provider.queue_response(content="ok")
_assert_parity(tmp_path, script, allowed_tools=["save_file"])
def test_a_rejected_command_behaves_identically(tmp_path: Path) -> None:
# The security-critical path: the gate says no, so the command must never
# run and the model must read back the same refusal in both designs.
def script(provider: FakeProvider) -> None:
provider.queue_response(
content="Running it.",
tool_calls=[{"id": "c1", "name": "run_command",
"arguments": {"command": "echo hi"}}],
)
provider.queue_response(content="Understood.")
_assert_parity(tmp_path, script, approve=False)
def test_hitting_the_step_ceiling_behaves_identically(tmp_path: Path) -> None:
# The model never stops calling tools, so both paths must stop at the same
# place and say so the same way.
def script(provider: FakeProvider) -> None:
for i in range(4):
provider.queue_response(
content=f"step {i}",
tool_calls=[{"id": f"c{i}", "name": "save_file",
"arguments": {"filename": f"f{i}.md", "content": "x"}}],
)
_assert_parity(tmp_path, script, max_steps=2)
+220
View File
@@ -0,0 +1,220 @@
"""R04-T04 — the migrated Cowork call site, exercised end to end without Qt.
``CoworkTab.build_job`` only ever *reads attributes* off its widget, so the real
production method can be invoked against a stand-in that supplies those
attributes. That is what happens here: the actual ``build_job`` body runs, builds
a request, wires the service through ``core_runtime_adapter``, and drives a real
turn (real tool execution, real output-folder cleanup) against ``FakeProvider``.
Why it matters: this is the only automated check that the widget's contract with
the service still holds — that the worker's list is appended to in place (the
transcript re-render and history merge both read it), that events still arrive as
legacy dicts, and that a produced file really lands in the turn's folder. None of
it needs a display server, so it runs in CI like every other test.
"""
from __future__ import annotations
import copy
from pathlib import Path
from typing import Any, Dict, List, Optional
import pytest
from cowork_local.config import DEFAULT_CONFIG, AppConfig
from cowork_local.tests.fakes.fake_provider import FakeProvider
class _FakeWorker:
"""The parts of ``core/worker.py::AgentWorker`` a job actually touches."""
def __init__(self, approve_commands: bool = True) -> None:
self.events: List[Dict[str, Any]] = []
self.gate: Optional[Any] = None
self._approve = approve_commands
self.cancelled = False
def emit_event(self, event: Dict[str, Any]) -> None:
self.events.append(event)
def is_cancelled(self) -> bool:
return self.cancelled
def new_gate(self, mode: str, agent_role: str = "") -> Any:
# Mirrors AgentWorker.new_gate: the gate is stored on the worker so the
# UI thread can resolve it, and answers request() from the worker thread.
worker = self
class _Gate:
requests: List[Dict[str, Any]] = []
def request(self, action: Dict[str, Any]) -> bool:
self.requests.append(action)
return worker._approve
self.gate = _Gate()
return self.gate
class _FakeCtx:
"""The ``AppContext`` surface ``build_job`` uses."""
def __init__(self, config: AppConfig, confirm_commands: bool = False) -> None:
self.config = config
self._confirm = confirm_commands
def project_confirm_commands(self) -> bool:
return self._confirm
def build_mcp_tools(self):
return [], None
class _WidgetStub:
"""Stands in for the CoworkTab instance ``build_job`` reads its state from."""
kind = "cowork"
def __init__(self, out_root: Path, ctx: _FakeCtx, provider: FakeProvider) -> None:
self._out_root = out_root
self.ctx = ctx
self._provider = provider
self.title = "Report"
self.session_id = "s1"
self.project_id = "" # the auto-seeded default workspace
self._model = ""
self._routed_provider = None
self._routed_model = None
def _session_output_dir(self) -> Path:
return self._out_root
def workspace_dir(self) -> Path:
return self._out_root
def admin_agent_prompt(self) -> str:
return ""
def build_provider(self) -> FakeProvider:
return self._provider
def _config() -> AppConfig:
"""A real AppConfig that never touches ``~/.cowork_local``.
The AI security guardrails are switched off: they would call the model to
review the prompt, which is a separate feature with its own tests and would
make this one depend on what the fake answers.
"""
data = copy.deepcopy(DEFAULT_CONFIG)
data["agent_security"]["enabled"] = False
return AppConfig(data)
def _run_turn(tmp_path: Path, provider: FakeProvider, messages: List[Dict[str, Any]],
*, confirm_commands: bool = False, approve: bool = True):
"""Invoke the real ``CoworkTab.build_job`` against the stub and run its job."""
from cowork_local.ui.cowork_tab import CoworkTab
out_dir = tmp_path / ".turns" / "t1"
out_dir.mkdir(parents=True, exist_ok=True)
widget = _WidgetStub(tmp_path, _FakeCtx(_config(), confirm_commands), provider)
worker = _FakeWorker(approve_commands=approve)
job = CoworkTab.build_job(widget, "make me a report", messages, out_dir)
result = job(worker)
return result, worker
def test_the_turn_runs_and_reports_its_folder(tmp_path: Path) -> None:
provider = FakeProvider()
provider.queue_response(content="Here you go.", chunks=["Here ", "you go."])
messages = [{"role": "user", "content": "make me a report"}]
result, worker = _run_turn(tmp_path, provider, messages)
assert result["turn_dir"] == str(tmp_path / ".turns" / "t1")
assert [e["type"] for e in worker.events] == [
"text", "text", "assistant_done", "turn_completed"]
def test_the_worker_list_is_appended_to_in_place(tmp_path: Path) -> None:
# _reattach_running_turn replays from this very list while the turn runs, and
# _finalize_turn slices it by the pre-turn length afterwards.
provider = FakeProvider()
provider.queue_response(content="Done.")
user = {"role": "user", "content": "make me a report"}
messages = [user]
result, _ = _run_turn(tmp_path, provider, messages)
assert result["messages"] is messages
# Identity, not just equality: _reattach_running_turn locates the turn's user
# message with ``m is ctx["user_msg"]`` to replay the steps after it.
assert any(m is user for m in messages)
# Several system blocks are expected — the tool prompt plus the tagged
# skills/security-rules blocks the runtime refreshes on every turn.
assert [m["role"] for m in messages if m["role"] != "system"] == ["user", "assistant"]
assert messages[-1]["content"] == "Done."
def test_a_saved_file_lands_in_the_turn_folder(tmp_path: Path) -> None:
provider = FakeProvider()
provider.queue_response(
content="Writing it.",
tool_calls=[{"id": "c1", "name": "save_file",
"arguments": {"filename": "report.md", "content": "# Report\n"}}],
)
provider.queue_response(content="Saved.")
messages = [{"role": "user", "content": "make me a report"}]
_, worker = _run_turn(tmp_path, provider, messages)
produced = list((tmp_path / ".turns" / "t1").glob("*.md"))
assert len(produced) == 1
assert produced[0].read_text(encoding="utf-8") == "# Report\n"
results = [e for e in worker.events if e["type"] == "tool_result"]
assert results and results[0]["ok"] is True
def test_auto_run_mode_never_creates_a_permission_gate(tmp_path: Path) -> None:
provider = FakeProvider()
provider.queue_response(content="ok")
_, worker = _run_turn(tmp_path, provider, [{"role": "user", "content": "hi"}],
confirm_commands=False)
assert worker.gate is None
def test_confirm_mode_creates_the_gate_and_a_refusal_stops_the_command(tmp_path: Path) -> None:
provider = FakeProvider()
provider.queue_response(
content="Running it.",
tool_calls=[{"id": "c1", "name": "run_command",
"arguments": {"command": "echo hi"}}],
)
provider.queue_response(content="Understood.")
_, worker = _run_turn(tmp_path, provider, [{"role": "user", "content": "run it"}],
confirm_commands=True, approve=False)
assert worker.gate is not None
refusals = [e for e in worker.events
if e["type"] == "tool_result" and e["output"] == "Rejected by user."]
assert len(refusals) == 1
def test_cancelling_before_the_turn_starts_calls_no_model(tmp_path: Path) -> None:
from cowork_local.ui.cowork_tab import CoworkTab
provider = FakeProvider()
provider.queue_response(content="never")
out_dir = tmp_path / ".turns" / "t1"
out_dir.mkdir(parents=True)
widget = _WidgetStub(tmp_path, _FakeCtx(_config()), provider)
worker = _FakeWorker()
worker.cancelled = True
CoworkTab.build_job(widget, "x", [{"role": "user", "content": "x"}], out_dir)(worker)
assert provider.call_count == 0
@@ -0,0 +1,191 @@
"""R04-T05 — the Schedule Task runner's cowork branch, pinned before and after.
Written against the CURRENT ``_run_agent`` first, as the safety net for moving it
onto ``ConversationApplicationService``: an unattended run has five behaviours the
interactive path does not have (the plan reminder prefixed to the prompt, the
session registered in History before the model starts, a re-save after every
assistant message, the timeout notice, and the "did the agent's own checklist
finish?" report), and none of them was covered by a test.
Everything is isolated from the user's real config: history goes to ``tmp_path``
via ``history.custom_dir`` and the AI guardrails are off, so no run touches
``~/.cowork_local`` or calls a model to review a prompt.
"""
from __future__ import annotations
import copy
import json
from pathlib import Path
from typing import Any, Dict, List, Optional
from cowork_local.config import DEFAULT_CONFIG, AppConfig
from cowork_local.core import task_executors
from cowork_local.tests.fakes.fake_provider import FakeProvider
class _FakeCtx:
"""The ``AppContext`` surface ``_run_agent`` touches."""
def __init__(self, config: AppConfig, provider: FakeProvider) -> None:
self.config = config
self._provider = provider
def build_active_provider(self) -> FakeProvider:
return self._provider
def build_provider_for(self, name=None, model=None) -> FakeProvider:
return self._provider
def _config(tmp_path: Path) -> AppConfig:
data = copy.deepcopy(DEFAULT_CONFIG)
# Keep the run entirely offline and off the real config dir.
data["agent_security"]["enabled"] = False
data["history"]["custom_dir"] = str(tmp_path / "history")
return AppConfig(data)
def _run(tmp_path: Path, provider: FakeProvider, *, prompt: str = "write the report",
timeout_sec: Optional[int] = None, admin_agent: Any = None):
"""Run one cowork task and return ``(result_tuple, events, config)``."""
out_dir = tmp_path / "run"
out_dir.mkdir(parents=True, exist_ok=True)
config = _config(tmp_path)
events: List[Dict[str, Any]] = []
result = task_executors._run_agent(
_FakeCtx(config, provider), "cowork", prompt, out_dir,
events.append, lambda: False, title="Weekly report",
timeout_sec=timeout_sec, admin_agent=admin_agent,
)
return result, events, config
def _saved_conversation(config: AppConfig) -> Dict[str, Any]:
"""The single conversation the run wrote into the isolated history folder."""
files = list(Path(config.history_dir()).rglob("*.json"))
assert len(files) == 1, f"expected one saved conversation, found {files}"
return json.loads(files[0].read_text(encoding="utf-8"))
# --------------------------------------------------------------------------- #
def test_a_cowork_task_returns_the_final_answer(tmp_path: Path) -> None:
provider = FakeProvider()
provider.queue_response(content="Report is ready.")
(answer, timed_out, incomplete), _events, _config = _run(tmp_path, provider)
assert answer == "Report is ready."
assert timed_out is False
assert incomplete == ""
def test_the_plan_reminder_is_prefixed_to_the_prompt(tmp_path: Path) -> None:
# An unattended run has nobody watching, so the agent is pushed to keep its
# own checklist honest. The reminder must lead the message.
provider = FakeProvider()
provider.queue_response(content="ok")
_run(tmp_path, provider, prompt="write the report")
sent = provider.call_history[0][-1]["content"]
assert sent.startswith("This runs unattended (Schedule Task)")
assert sent.endswith("write the report")
def test_an_admin_agent_persona_sits_between_the_reminder_and_the_prompt(
tmp_path: Path) -> None:
class _Agent:
# An admin agent may pin its own provider/model; blank means "use the
# machine's Settings default", which is what build_agent_provider reads.
provider = ""
model = ""
def effective_prompt(self) -> str:
return "You are the reporting agent."
provider = FakeProvider()
provider.queue_response(content="ok")
_run(tmp_path, provider, prompt="write the report", admin_agent=_Agent())
sent = provider.call_history[0][-1]["content"]
assert sent.index("This runs unattended") < sent.index("You are the reporting agent.")
assert sent.index("You are the reporting agent.") < sent.index("write the report")
def test_the_session_is_announced_once_it_exists_on_disk(tmp_path: Path) -> None:
# The scheduler refreshes History on this event, so it must not fire before
# the conversation is really there.
provider = FakeProvider()
provider.queue_response(content="ok")
_result, events, config = _run(tmp_path, provider)
ready = [e for e in events if e["type"] == "history_ready"]
assert len(ready) == 1
assert ready[0]["session_id"]
assert _saved_conversation(config)["session_id"] == ready[0]["session_id"]
def test_the_saved_conversation_carries_the_answer_and_the_task_title(
tmp_path: Path) -> None:
provider = FakeProvider()
provider.queue_response(content="Report is ready.")
_result, _events, config = _run(tmp_path, provider)
saved = _saved_conversation(config)
assert saved["title"] == "[Task] Weekly report"
assert saved["messages"][-1] == {"role": "assistant", "content": "Report is ready."}
def test_an_unfinished_checklist_is_reported_back_to_the_scheduler(
tmp_path: Path) -> None:
# The agent ticked no step to done, so the task must not be called finished
# just because no exception was raised.
provider = FakeProvider()
provider.queue_response(
content="Working on it.",
tool_calls=[{"id": "c1", "name": "update_plan",
"arguments": {"steps": [{"title": "Draft", "status": "running"}]}}],
)
provider.queue_response(content="Stopping here.")
(_answer, _timed_out, incomplete), _events, _config = _run(tmp_path, provider)
assert incomplete
assert "Draft" in incomplete
def test_a_finished_checklist_reports_nothing_outstanding(tmp_path: Path) -> None:
provider = FakeProvider()
provider.queue_response(
content="Done.",
tool_calls=[{"id": "c1", "name": "update_plan",
"arguments": {"steps": [{"title": "Draft", "status": "done"}]}}],
)
provider.queue_response(content="All done.")
(_answer, _timed_out, incomplete), _events, _config = _run(tmp_path, provider)
assert incomplete == ""
def test_running_out_of_time_appends_the_timeout_notice_to_the_conversation(
tmp_path: Path) -> None:
# A negative timeout puts the deadline in the past, which is the only
# deterministic way to exercise a wall-clock branch in a unit test.
provider = FakeProvider()
provider.queue_response(content="never gets there")
(answer, timed_out, incomplete), events, config = _run(
tmp_path, provider, timeout_sec=-1)
assert timed_out is True
assert incomplete == "" # a timeout is not an unfinished checklist
assert "quá thời gian chờ" in answer
assert any(e["type"] == "assistant_done" and "quá thời gian chờ" in e["content"]
for e in events)
assert "quá thời gian chờ" in _saved_conversation(config)["messages"][-1]["content"]
@@ -0,0 +1,293 @@
"""R04-T03 (b) — the turn loop: composition, tool dispatch, budget, cancel.
Behaviour that used to be reachable only by running the real widget. Every
dependency is a fake from ``tests/fakes/turn_runtime_fakes.py``, so the file
runs in milliseconds and each test states one rule of the loop.
"""
from __future__ import annotations
from typing import Any, Dict, List, Tuple
from cowork_local.application.conversations.conversation_application_service import (
ConversationApplicationService,
)
from cowork_local.domain.agents.agent_event import (
AssistantMessageCompletedEvent,
PlanStep,
PlanUpdatedEvent,
TextChunkEvent,
ToolCallFinishedEvent,
ToolCallStartedEvent,
ToolOutputChunkEvent,
ToolPreview,
TurnCompletedEvent,
)
from cowork_local.tests.fakes.turn_runtime_fakes import (
FakeModelCall,
FakeReply,
FakeToolRuntime,
events_of_type,
make_request,
run_turn,
tool_turn,
)
def _service(model, tools, **overrides) -> ConversationApplicationService:
return ConversationApplicationService(model, tools, **overrides)
# --------------------------------------------------------------------------- #
# The happy path.
# --------------------------------------------------------------------------- #
def test_a_plain_answer_streams_text_then_reports_the_message_and_the_turn() -> None:
model = FakeModelCall([FakeReply(content="Hello there", chunks=["Hello ", "there"])])
result, events = run_turn(_service(model, FakeToolRuntime()))
assert [e.delta for e in events_of_type(events, TextChunkEvent)] == ["Hello ", "there"]
assert events_of_type(events, AssistantMessageCompletedEvent) == [
AssistantMessageCompletedEvent(content="Hello there")]
assert events_of_type(events, TurnCompletedEvent) == [
TurnCompletedEvent(final_text="Hello there", steps_used=1)]
assert result.final_text == "Hello there"
assert result.ok is True
def test_the_composed_user_message_is_appended_before_the_first_call() -> None:
model = FakeModelCall([FakeReply(content="ok")])
request = make_request(prompt="ship it", instruction_prefix="RULES",
session_notes="earlier: a.md",
messages=[{"role": "user", "content": "previous"}])
run_turn(_service(model, FakeToolRuntime()), request)
sent = model.calls[0]["messages"]
assert sent[-1] == {"role": "user",
"content": "RULES\n\n---\n\nship it\n\nearlier: a.md"}
assert sent[-2] == {"role": "user", "content": "previous"}
def test_attachments_are_read_when_the_turn_runs_not_when_it_was_built() -> None:
# Extraction can pip-install a parser or shell out to LibreOffice, so it must
# happen here (worker thread), not while the UI was assembling the request.
seen: List[Tuple[str, Tuple[str, ...]]] = []
def reader(prompt: str, attachments: Tuple[str, ...]) -> str:
seen.append((prompt, attachments))
return f"{prompt}\n\n<contents of {len(attachments)} file(s)>"
model = FakeModelCall([FakeReply(content="ok")])
request = make_request(prompt="summarise", attachments=["a.docx", "b.pdf"])
run_turn(_service(model, FakeToolRuntime(), attachment_reader=reader), request)
assert seen == [("summarise", ("a.docx", "b.pdf"))]
assert "contents of 2 file(s)" in model.calls[0]["messages"][-1]["content"]
def test_the_prompt_preparer_is_told_which_tools_the_turn_advertises() -> None:
# The system prompt gains an MS365 paragraph only when ms365__* tools are
# present, so the preparer has to see the real list.
seen: List[Tuple[str, ...]] = []
model = FakeModelCall([FakeReply(content="ok")])
tools = FakeToolRuntime(specs=("save_file", "ms365__send_mail"))
run_turn(_service(model, tools,
prepare_prompt=lambda messages, names: seen.append(names)))
assert seen == [("save_file", "ms365__send_mail")]
def test_only_the_allowed_tools_are_advertised() -> None:
model = FakeModelCall([FakeReply(content="ok")])
tools = FakeToolRuntime(specs=("save_file", "run_command", "update_plan"))
run_turn(_service(model, tools), make_request(allowed_tools=("save_file", "update_plan")))
assert model.calls[0]["tool_names"] == ["save_file", "update_plan"]
# --------------------------------------------------------------------------- #
# Tool dispatch.
# --------------------------------------------------------------------------- #
def tool_turn(tool_name: str = "save_file", args=None, **tool_kwargs):
"""A turn that calls one tool, then answers."""
calls = [{"id": "c1", "name": tool_name, "arguments": args or {"filename": "a.md"}}]
model = FakeModelCall([FakeReply(content="working", tool_calls=calls),
FakeReply(content="done")])
return model, FakeToolRuntime(**tool_kwargs)
def test_a_tool_call_is_announced_executed_and_answered_in_the_message_list() -> None:
model, tools = tool_turn(results={"save_file": {"ok": True, "output": "saved",
"path": "out/a.md"}})
result, events = run_turn(_service(model, tools))
assert events_of_type(events, ToolCallStartedEvent) == [ToolCallStartedEvent(
call_id="c1", name="save_file", arguments={"filename": "a.md"},
preview=ToolPreview(kind="info", title="save_file", text="{'filename': 'a.md'}"))]
assert events_of_type(events, ToolCallFinishedEvent) == [ToolCallFinishedEvent(
call_id="c1", name="save_file", ok=True, output="saved", path="out/a.md")]
assert tools.executed == [("save_file", {"filename": "a.md"})]
assert result.messages[-2] == {"role": "tool", "tool_call_id": "c1",
"name": "save_file", "content": "saved"}
def test_live_tool_output_is_streamed_while_the_tool_runs() -> None:
model, tools = tool_turn("run_command", {"command": "ls"})
tools.emit_output = "file-a\n"
_, events = run_turn(_service(model, tools))
assert events_of_type(events, ToolOutputChunkEvent) == [ToolOutputChunkEvent(
call_id="c1", name="run_command", delta="file-a\n")]
def test_the_loop_ends_as_soon_as_the_model_stops_calling_tools() -> None:
model, tools = tool_turn()
result, _ = run_turn(_service(model, tools))
assert result.steps_used == 2
assert result.budget_exhausted is False
def test_the_plan_tool_reports_a_plan_update_and_no_tool_bubble() -> None:
calls = [{"id": "c1", "name": "update_plan",
"arguments": {"steps": [{"title": "Draft", "status": "running"}]}}]
model = FakeModelCall([FakeReply(content="planning", tool_calls=calls), FakeReply(content="done")])
tools = FakeToolRuntime(results={"update_plan": {
"ok": True, "output": "Plan updated.",
"plan_steps": [PlanStep(title="Draft", status="running")]}})
result, events = run_turn(_service(model, tools))
assert events_of_type(events, PlanUpdatedEvent) == [
PlanUpdatedEvent(steps=(PlanStep(title="Draft", status="running"),))]
assert events_of_type(events, ToolCallStartedEvent) == []
assert events_of_type(events, ToolCallFinishedEvent) == []
assert result.plan_steps == (PlanStep(title="Draft", status="running"),)
# --------------------------------------------------------------------------- #
# Budget, cancellation.
# --------------------------------------------------------------------------- #
def test_running_out_of_steps_is_flagged_and_announced() -> None:
# The model keeps calling tools forever; the ceiling must stop it visibly.
forever = [FakeReply(content=f"step {i}",
tool_calls=[{"id": f"c{i}", "name": "save_file", "arguments": {}}])
for i in range(5)]
model = FakeModelCall(forever)
result, events = run_turn(_service(model, FakeToolRuntime()), make_request(max_steps=2))
assert result.steps_used == 2
assert result.budget_exhausted is True
assert "2-step safety limit" in events_of_type(events, TextChunkEvent)[-1].delta
# The note reaches the transcript but NOT the stored answer: a turn that hits
# the ceiling always ends on a tool message, and the existing runtime only
# merges the note when the last message is the assistant's. Pinned here so a
# future change to that rule is a deliberate decision, not a silent drift.
assert result.final_text == "step 1"
def test_run_to_completion_uses_the_higher_ceiling() -> None:
forever = [FakeReply(content="x", tool_calls=[{"id": "c", "name": "save_file", "arguments": {}}])
for _ in range(6)]
model = FakeModelCall(forever)
result, _ = run_turn(_service(model, FakeToolRuntime()),
make_request(max_steps=2, completion_max_steps=5, run_to_completion=True))
assert result.steps_used == 5
def test_a_turn_cancelled_before_it_starts_never_calls_the_model() -> None:
model = FakeModelCall([FakeReply(content="never")])
result, events = run_turn(_service(model, FakeToolRuntime()), cancel=lambda: True)
assert model.calls == []
assert result.cancelled is True
assert result.budget_exhausted is False
assert events_of_type(events, TurnCompletedEvent) == [TurnCompletedEvent(cancelled=True)]
def test_cancelling_during_a_turn_stops_dispatching_the_remaining_tool_calls() -> None:
calls = [{"id": "c1", "name": "save_file", "arguments": {}},
{"id": "c2", "name": "save_file", "arguments": {}}]
model = FakeModelCall([FakeReply(content="two tools", tool_calls=calls)])
tools = FakeToolRuntime()
stop = {"now": False}
def cancel() -> bool:
return stop["now"]
original_execute = tools.execute
def execute(name, args, on_output=None, cancel=None):
stop["now"] = True # cancel raised while the first tool runs
return original_execute(name, args, on_output=on_output, cancel=cancel)
tools.execute = execute
result, _ = run_turn(_service(model, tools), cancel=cancel)
assert len(tools.executed) == 1
assert result.cancelled is True
# --------------------------------------------------------------------------- #
# Bring-your-own working list.
#
# ``ui/chat_panel.py`` holds the turn's message list in its own turn context and
# reads it WHILE the worker appends (``_reattach_running_turn`` replays the steps
# done so far when the user reopens a running conversation; ``_finalize_turn``
# slices it by ``snapshot_len``). A service that built its own private list would
# silently break both, so a caller can hand its list over instead.
# --------------------------------------------------------------------------- #
def test_a_caller_supplied_list_is_appended_to_in_place() -> None:
model, tools = tool_turn()
live: List[Dict[str, Any]] = [{"role": "user", "content": "already composed"}]
result = ConversationApplicationService(model, tools).execute(
make_request(), lambda event: None, messages=live)
roles = [m["role"] for m in live]
assert roles == ["user", "assistant", "tool", "assistant"]
assert result.messages == tuple(live)
def test_a_caller_supplied_list_is_used_as_is_without_recomposing_the_prompt() -> None:
# The widget already applied the skill prefix and the session notes when it
# built its message; composing again would duplicate them.
model = FakeModelCall([FakeReply(content="ok")])
user = {"role": "user", "content": "already composed"}
live = [user]
ConversationApplicationService(model, FakeToolRuntime()).execute(
make_request(prompt="typed text", instruction_prefix="RULES",
session_notes="notes"),
lambda event: None, messages=live)
assert live[0] is user
assert live[0]["content"] == "already composed"
assert [m["role"] for m in live].count("user") == 1
def test_a_caller_supplied_list_skips_the_attachment_reader() -> None:
# Reading the attachments is what produced the caller's message in the first
# place; doing it again would re-parse every file.
model = FakeModelCall([FakeReply(content="ok")])
calls: List[Any] = []
ConversationApplicationService(
model, FakeToolRuntime(),
attachment_reader=lambda prompt, attachments: calls.append(prompt) or prompt,
).execute(make_request(attachments=["a.docx"]), lambda event: None,
messages=[{"role": "user", "content": "composed"}])
assert calls == []
+210
View File
@@ -0,0 +1,210 @@
"""R04-T03 (b) — the turn loop: guards, permission gate, compaction, cleanup.
Split out of ``test_conversation_application_service.py`` to keep each file
inside the 400-LOC limit. Same fakes, same service; this half pins the ORDER of
the safety steps (guard before model, guard before execute, gate before execute)
and the promise that the output sandbox is tidied on the way out.
"""
from __future__ import annotations
from typing import Any, Dict, List
import pytest
from cowork_local.application.conversations.conversation_application_service import (
ConversationApplicationService,
)
from cowork_local.domain.agents.agent_event import (
ErrorEvent,
OutputsAddedEvent,
ReasoningChunkEvent,
TextChunkEvent,
ToolCallFinishedEvent,
)
from cowork_local.tests.fakes.turn_runtime_fakes import (
FakeModelCall,
FakeReply,
FakeToolRuntime,
events_of_type,
make_request,
run_turn,
tool_turn,
)
def _service(model, tools, **overrides) -> ConversationApplicationService:
return ConversationApplicationService(model, tools, **overrides)
# --------------------------------------------------------------------------- #
# Guards and the permission gate.
# --------------------------------------------------------------------------- #
def test_the_prompt_guard_runs_before_the_model_is_ever_called() -> None:
order: List[str] = []
model = FakeModelCall([FakeReply(content="ok")])
model_call = model.call
def call(*a, **kw):
order.append("model")
return model_call(*a, **kw)
model.call = call
run_turn(_service(model, FakeToolRuntime(), prompt_guard=lambda messages: order.append("guard")))
assert order == ["guard", "model"]
def test_a_blocked_prompt_propagates_before_the_output_folder_is_touched() -> None:
model = FakeModelCall([FakeReply(content="never")])
tools = FakeToolRuntime()
events: List[Any] = []
def guard(messages) -> None:
raise RuntimeError("SecurityBlocked: nope")
service = _service(model, tools, prompt_guard=guard)
with pytest.raises(RuntimeError, match="SecurityBlocked"):
service.execute(make_request(), events.append)
assert model.calls == []
assert events_of_type(events, ErrorEvent) == [ErrorEvent(message="SecurityBlocked: nope")]
# Cleanup is NOT a read-only operation (it deletes a stale .scratch and every
# empty sub-folder), so a turn rejected before it started must not run it.
assert tools.finalize_calls == []
def test_output_cleanup_still_runs_when_the_turn_fails_mid_loop() -> None:
# Once the turn has started producing files, the sandbox must be tidied on
# the way out no matter how the turn ends.
model = FakeModelCall([RuntimeError("gateway exploded")])
tools = FakeToolRuntime()
events: List[Any] = []
with pytest.raises(RuntimeError, match="gateway exploded"):
_service(model, tools).execute(make_request(), events.append)
assert tools.finalize_calls == [{"before": "before", "cancelled": False}]
assert events_of_type(events, ErrorEvent) == [ErrorEvent(message="gateway exploded")]
def test_the_command_guard_runs_before_the_tool_executes() -> None:
order: List[str] = []
model, tools = tool_turn("run_command", {"command": "ls"})
original = tools.execute
def execute(name, args, on_output=None, cancel=None):
order.append("execute")
return original(name, args, on_output=on_output, cancel=cancel)
tools.execute = execute
run_turn(_service(model, tools,
command_guard=lambda name, args: order.append(f"guard:{name}")))
assert order == ["guard:run_command", "execute"]
def test_disabling_rule_enforcement_skips_both_guards() -> None:
# Co4E flow steps run inside the workspace sandbox and opt out on purpose.
calls: List[str] = []
model, tools = tool_turn("run_command", {"command": "ls"})
run_turn(_service(model, tools,
prompt_guard=lambda messages: calls.append("prompt"),
command_guard=lambda name, args: calls.append("command")),
make_request(enforce_rules=False))
assert calls == []
def test_the_permission_gate_is_asked_only_for_command_tools() -> None:
asked: List[str] = []
model, tools = tool_turn("save_file", {"filename": "a.md"})
run_turn(_service(model, tools,
permission_request=lambda action: asked.append(action["name"]) or True),
make_request(gate_mode="confirm"))
assert asked == [] # save_file writes into the sandbox: never gated
def test_a_command_tool_in_confirm_mode_asks_before_running() -> None:
asked: List[Dict[str, Any]] = []
model, tools = tool_turn("run_command", {"command": "ls"})
def approve(action: Dict[str, Any]) -> bool:
asked.append(action)
return True
run_turn(_service(model, tools, permission_request=approve), make_request(gate_mode="confirm"))
assert [a["name"] for a in asked] == ["run_command"]
assert tools.executed == [("run_command", {"command": "ls"})]
def test_a_rejected_command_is_reported_as_a_failed_tool_and_never_runs() -> None:
model, tools = tool_turn("run_command", {"command": "rm -rf /"})
result, events = run_turn(_service(model, tools, permission_request=lambda action: False),
make_request(gate_mode="confirm"))
assert tools.executed == []
assert events_of_type(events, ToolCallFinishedEvent) == [ToolCallFinishedEvent(
call_id="c1", name="run_command", ok=False, output="Rejected by user.")]
assert result.messages[-2]["content"] == "Rejected by user."
def test_auto_mode_never_asks_even_for_a_command() -> None:
model, tools = tool_turn("run_command", {"command": "ls"})
def refuse(action): # would block the turn if it were consulted
raise AssertionError("the gate must not be consulted in auto mode")
run_turn(_service(model, tools, permission_request=refuse), make_request(gate_mode="auto"))
assert tools.executed == [("run_command", {"command": "ls"})]
# --------------------------------------------------------------------------- #
# Context compaction, reasoning, output cleanup.
# --------------------------------------------------------------------------- #
def test_the_conversation_is_offered_for_compaction_before_every_call() -> None:
compactions: List[int] = []
model, tools = tool_turn()
run_turn(_service(model, tools,
compact=lambda messages, cancel: compactions.append(len(messages))))
assert len(compactions) == 2 # once per provider call
def test_reasoning_is_streamed_as_its_own_event() -> None:
model = FakeModelCall([FakeReply(content="42", reasoning="thinking...")])
_, events = run_turn(_service(model, FakeToolRuntime()))
assert events_of_type(events, ReasoningChunkEvent) == [ReasoningChunkEvent(delta="thinking...")]
def test_a_reasoning_only_reply_gets_a_visible_note_in_the_transcript() -> None:
# Otherwise a Schedule Task run reads back an empty answer and writes
# "(no output)" into its report.
model = FakeModelCall([FakeReply(content="", reasoning="thought hard")])
result, events = run_turn(_service(model, FakeToolRuntime()))
assert "only its reasoning" in events_of_type(events, TextChunkEvent)[-1].delta
assert "only its reasoning" in result.final_text
def test_promoted_and_discarded_output_files_are_reported_at_the_end() -> None:
model = FakeModelCall([FakeReply(content="ok")])
tools = FakeToolRuntime(added=("out/report.pptx",))
_, events = run_turn(_service(model, tools))
assert events_of_type(events, OutputsAddedEvent) == [
OutputsAddedEvent(paths=("out/report.pptx",))]
assert tools.finalize_calls == [{"before": "before", "cancelled": False}]
+76
View File
@@ -0,0 +1,76 @@
"""R04-T04 — unit tests for the UI-state -> request mapping.
Three small rules used to sit inline in ``ui/cowork_tab.py::build_job``, where no
test could reach them: the turn's prompt is the last message in the working list,
the history is everything before it, and the confirm-commands flag becomes a gate
mode. Getting any of them wrong is silent (a duplicated user message, a command
that stops asking for approval), so they are pinned here.
"""
from __future__ import annotations
from pathlib import Path
from cowork_local.application.conversations.cowork_turn_request import (
build_cowork_turn_request,
)
def _build(**overrides):
base = {
"turn_id": "t3",
"session_id": "s1",
"messages": [{"role": "user", "content": "make me a report"}],
}
base.update(overrides)
return build_cowork_turn_request(**base)
def test_the_last_message_becomes_the_prompt_and_the_rest_the_history() -> None:
request = _build(messages=[
{"role": "user", "content": "earlier"},
{"role": "assistant", "content": "sure"},
{"role": "user", "content": "now this"},
])
assert request.prompt == "now this"
assert request.messages == ({"role": "user", "content": "earlier"},
{"role": "assistant", "content": "sure"})
def test_an_empty_working_list_yields_an_empty_prompt() -> None:
# Defensive: a turn with no message at all must not raise on messages[-1].
request = _build(messages=[])
assert request.prompt == ""
assert request.messages == ()
def test_confirming_commands_puts_the_turn_in_confirm_gate_mode() -> None:
assert _build(confirm_commands=True).gate_mode == "confirm"
assert _build(confirm_commands=False).gate_mode == "auto"
assert _build().gate_mode == "auto" # auto-run is the default
def test_the_captured_widget_state_is_carried_into_the_request() -> None:
request = _build(
surface="cowork", project_id="p7", title="Weekly report",
provider_id="anthropic", model="claude-sonnet-4-6",
instructions="PROJECT RULES", output_dir="out/.turns/t3",
home_output_root="out", agent_role="cowork",
)
assert (request.turn_id, request.session_id) == ("t3", "s1")
assert (request.surface, request.project_id, request.title) == \
("cowork", "p7", "Weekly report")
assert (request.provider_id, request.model) == ("anthropic", "claude-sonnet-4-6")
assert request.project_context == "PROJECT RULES"
assert request.output_dir == Path("out/.turns/t3")
assert request.home_output_root == Path("out")
assert request.agent_role == "cowork"
def test_the_prompt_survives_a_message_whose_content_is_missing() -> None:
request = _build(messages=[{"role": "user"}])
assert request.prompt == ""
+34
View File
@@ -0,0 +1,34 @@
"""R04-T05 — unit tests for the unattended-run prompt assembly.
``_run_agent`` used to build this by rebinding ``prompt`` three times, each with
its own ``f"{block}\n\n{prompt}"``. The ORDER that produced is load-bearing (the
plan reminder has to lead, the task's own words have to trail) and it was
readable only by replaying the rebindings in your head.
"""
from __future__ import annotations
from cowork_local.core.task_executors import _unattended_prompt
def test_the_plan_reminder_leads_and_the_task_prompt_trails() -> None:
built = _unattended_prompt("write the report")
assert built.startswith("This runs unattended (Schedule Task)")
assert built.endswith("write the report")
def test_a_skill_block_sits_between_the_reminder_and_the_agent_persona() -> None:
built = _unattended_prompt("write the report", skill_text="SKILL",
agent_instructions="PERSONA")
assert built.index("This runs unattended") < built.index("SKILL")
assert built.index("SKILL") < built.index("PERSONA")
assert built.index("PERSONA") < built.index("write the report")
def test_absent_blocks_leave_no_extra_blank_lines() -> None:
built = _unattended_prompt("do it", skill_text="", agent_instructions=None)
assert "\n\n\n" not in built
assert built.count("do it") == 1
+36
View File
@@ -0,0 +1,36 @@
"""R04-T04 — unit tests for the shared turn-runtime helpers.
``combine_instructions`` is the small rule the UI applied inline: a turn's
standing instructions are several independent blocks (project context, an Admin
agent's persona, a skill's rules, an unattended-run reminder) that must be joined
with one blank line, skipping whatever is absent. Two call sites need it (T04's
widget and T05's task runner), which is exactly when a rule stops being an inline
expression.
"""
from __future__ import annotations
from cowork_local.application.conversations.turn_runtime import combine_instructions
def test_two_blocks_are_joined_by_a_blank_line() -> None:
assert combine_instructions("PROJECT", "AGENT") == "PROJECT\n\nAGENT"
def test_an_absent_block_leaves_no_blank_line_behind() -> None:
assert combine_instructions("", "AGENT") == "AGENT"
assert combine_instructions("PROJECT", "") == "PROJECT"
assert combine_instructions("PROJECT", None) == "PROJECT"
def test_whitespace_only_blocks_do_not_count_as_instructions() -> None:
assert combine_instructions(" \n ", "AGENT") == "AGENT"
def test_nothing_to_say_produces_an_empty_string() -> None:
assert combine_instructions() == ""
assert combine_instructions("", None, " ") == ""
def test_more_than_two_blocks_keep_their_order() -> None:
assert combine_instructions("A", "B", "C") == "A\n\nB\n\nC"