feat(R04): run every Cowork turn through ConversationApplicationService
R04-T03 — the turn lifecycle, extracted from `core/chat_agent.py::run_cowork` into `application/conversations/`. The 260-line body mixed the lifecycle (step budget, cancel checks, guard -> preview -> gate -> execute ordering, sandbox tidy-up) with the machinery doing each step, and reaching any of it meant standing up a Qt widget and a worker thread. It is now a plain object driven through two Protocols and six callables (`turn_runtime.py`), with the concrete `core/*` wiring confined to `core_runtime_adapter.py` — the same shape R03 used for routing. Faithful port, not an improvement pass: where the original had a quirk (the step-ceiling note only merges into the answer when the last message is the assistant's) the quirk is preserved and commented. R04-T04 — `ui/cowork_tab.py::build_job` no longer calls run_cowork. It captures the widget's state at submit time, builds the request via the new `cowork_turn_request.py` and executes it. `execute(..., messages=...)` hands the widget's own list over because `_reattach_running_turn` replays from it WHILE the worker appends and `_finalize_turn` slices it afterwards — a private list would break both silently. R04-T05 — `core/task_executors.py`'s cowork branch shares the same engine. All five unattended-run behaviours stay put (plan reminder, history_ready, History autosave per assistant message, timeout notice, plan_incomplete_reason), and `_unattended_prompt` now expresses the load-bearing prefix order in one readable call instead of three successive rebindings. Verification: 74 new tests (364 passed, 1 skipped overall; check_imports PASS). The two that matter most: - `test_conversation_service_parity.py` runs the same scripted turn through run_cowork AND the service and compares the event stream, the resulting conversation and the advertised tool list across 7 scenarios; - `test_task_executor_turn.py` was written BEFORE the migration and passed 8/8 against the old code, then unchanged against the new. Known: `ui/cowork_tab.py` (416 -> 455) and `core/task_executors.py` (476 -> 524) stay above the 400-LOC limit. Both were already over it before this change; bringing them under needs the R08 / R07 decompositions. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,141 @@
|
||||
"""Offline test doubles for the R04 turn runtime seams.
|
||||
|
||||
Sits beside ``fake_provider.py``/``fake_tool_executor.py`` (R01-T02) and plays
|
||||
the same role one level up: those fake a *provider*, these fake the ports
|
||||
``ConversationApplicationService`` is driven through
|
||||
(``application/conversations/turn_runtime.py``).
|
||||
|
||||
Deliberately dumb — they record what they were asked and return canned answers.
|
||||
A failing test then points at the service under test rather than at a mock
|
||||
framework's configuration.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
from cowork_local.domain.agents.agent_event import ToolPreview
|
||||
from cowork_local.domain.agents.conversation_execution_request import (
|
||||
ConversationExecutionRequest,
|
||||
)
|
||||
|
||||
|
||||
class FakeSpec:
|
||||
"""An advertised tool. The service only ever reads ``.name`` off a spec."""
|
||||
|
||||
def __init__(self, name: str) -> None:
|
||||
self.name = name
|
||||
|
||||
|
||||
class FakeReply:
|
||||
"""One programmed provider answer."""
|
||||
|
||||
def __init__(self, content: str = "", tool_calls=None, chunks=None, reasoning: str = ""):
|
||||
self.content = content
|
||||
self.tool_calls = tool_calls or []
|
||||
# Default to streaming the whole content as a single chunk, which is what
|
||||
# a non-streaming gateway effectively does.
|
||||
self.chunks = chunks if chunks is not None else ([content] if content else [])
|
||||
self.reasoning = reasoning
|
||||
|
||||
|
||||
class FakeModelCall:
|
||||
""":class:`ModelCallPort` returning programmed replies in order.
|
||||
|
||||
A programmed entry may be an exception instead of a reply, which is how a
|
||||
test simulates the gateway dying mid-turn.
|
||||
"""
|
||||
|
||||
def __init__(self, replies: List[Any]) -> None:
|
||||
self.replies = list(replies)
|
||||
self.calls: List[Dict[str, Any]] = []
|
||||
|
||||
def call(self, messages, tools, on_text=None, on_reasoning=None, cancel=None):
|
||||
# Snapshot the messages: the service keeps mutating its own list, so
|
||||
# storing it by reference would make every recorded call look identical.
|
||||
self.calls.append({"messages": [dict(m) for m in messages],
|
||||
"tool_names": [getattr(t, "name", "") for t in tools]})
|
||||
reply = self.replies.pop(0) if self.replies else FakeReply(content="(default)")
|
||||
if isinstance(reply, BaseException):
|
||||
raise reply
|
||||
if reply.reasoning and on_reasoning:
|
||||
on_reasoning(reply.reasoning)
|
||||
for chunk in reply.chunks:
|
||||
if on_text and chunk:
|
||||
on_text(chunk)
|
||||
assistant: Dict[str, Any] = {"role": "assistant", "content": reply.content}
|
||||
if reply.tool_calls:
|
||||
assistant["tool_calls"] = reply.tool_calls
|
||||
return assistant
|
||||
|
||||
|
||||
class FakeToolRuntime:
|
||||
""":class:`ToolRuntimePort` over an imaginary output folder."""
|
||||
|
||||
def __init__(self, specs=("save_file", "run_command", "update_plan"),
|
||||
results: Optional[Dict[str, Dict[str, Any]]] = None,
|
||||
removed: Tuple[str, ...] = (), added: Tuple[str, ...] = ()) -> None:
|
||||
self._specs = [FakeSpec(n) for n in specs]
|
||||
self._results = results or {}
|
||||
self._removed, self._added = removed, added
|
||||
self.executed: List[Tuple[str, Dict[str, Any]]] = []
|
||||
self.finalize_calls: List[Dict[str, Any]] = []
|
||||
# When set, every executed tool streams this string through ``on_output``.
|
||||
self.emit_output: Optional[str] = None
|
||||
|
||||
def specs(self, allowed_tools=None):
|
||||
if allowed_tools is None:
|
||||
return list(self._specs)
|
||||
return [s for s in self._specs if s.name in allowed_tools]
|
||||
|
||||
def preview(self, name, args):
|
||||
return ToolPreview(kind="info", title=name, text=str(args))
|
||||
|
||||
def execute(self, name, args, on_output=None, cancel=None):
|
||||
self.executed.append((name, dict(args)))
|
||||
if self.emit_output and on_output:
|
||||
on_output(self.emit_output)
|
||||
return dict(self._results.get(name, {"ok": True, "output": f"{name} ok"}))
|
||||
|
||||
def snapshot(self):
|
||||
return "before"
|
||||
|
||||
def finalize(self, before, cancelled=False):
|
||||
self.finalize_calls.append({"before": before, "cancelled": cancelled})
|
||||
return list(self._removed), list(self._added)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Small helpers shared by the turn tests.
|
||||
# --------------------------------------------------------------------------- #
|
||||
def make_request(**overrides) -> ConversationExecutionRequest:
|
||||
"""A minimal valid request; each test overrides only what it exercises."""
|
||||
base: Dict[str, Any] = {"turn_id": "t1", "session_id": "s1", "prompt": "do it"}
|
||||
base.update(overrides)
|
||||
return ConversationExecutionRequest(**base)
|
||||
|
||||
|
||||
def run_turn(service, request=None, cancel=None):
|
||||
"""Execute a turn and return ``(result, events)``."""
|
||||
events: List[Any] = []
|
||||
result = service.execute(request or make_request(), events.append, cancel=cancel)
|
||||
return result, events
|
||||
|
||||
|
||||
def events_of_type(events, cls):
|
||||
"""Every emitted event of one type, in order."""
|
||||
return [e for e in events if isinstance(e, cls)]
|
||||
|
||||
|
||||
def tool_turn(tool_name: str = "save_file", args=None, **tool_kwargs):
|
||||
"""A turn that calls one tool and then answers — ``(model, tools)``."""
|
||||
calls = [{"id": "c1", "name": tool_name, "arguments": args or {"filename": "a.md"}}]
|
||||
model = FakeModelCall([FakeReply(content="working", tool_calls=calls),
|
||||
FakeReply(content="done")])
|
||||
return model, FakeToolRuntime(**tool_kwargs)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"FakeSpec", "FakeReply", "FakeModelCall", "FakeToolRuntime",
|
||||
"make_request", "run_turn", "events_of_type", "tool_turn",
|
||||
]
|
||||
@@ -0,0 +1,207 @@
|
||||
"""R04-T03 (c) — the service must behave exactly like ``run_cowork``.
|
||||
|
||||
The unit tests prove the loop follows the rules I wrote down. They cannot prove
|
||||
those rules are the ones the shipped runtime actually follows. This file does:
|
||||
each test scripts one provider, runs the SAME turn twice — once through
|
||||
``core/chat_agent.py::run_cowork``, once through
|
||||
``ConversationApplicationService`` wired by ``core_runtime_adapter`` — and
|
||||
compares the emitted event stream, the resulting conversation and the tool list
|
||||
the model was shown.
|
||||
|
||||
Anything the port got wrong (a missing event, a reordered guard, a different
|
||||
tool set, a changed message) fails here rather than in front of a user. The only
|
||||
allowed difference is the extra ``turn_completed`` event R04 introduces, which
|
||||
has no legacy consumer.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
from cowork_local.application.conversations.core_runtime_adapter import (
|
||||
build_cowork_conversation_service,
|
||||
legacy_event_sink,
|
||||
)
|
||||
from cowork_local.core import chat_agent
|
||||
from cowork_local.domain.agents.conversation_execution_request import (
|
||||
ConversationExecutionRequest,
|
||||
)
|
||||
from cowork_local.tests.fakes.fake_provider import FakeProvider
|
||||
|
||||
_USER_TURN = [{"role": "user", "content": "make me a report"}]
|
||||
|
||||
|
||||
class _FakeGate:
|
||||
"""Stands in for ``core/permissions.py::PermissionGate``."""
|
||||
|
||||
def __init__(self, approve: bool) -> None:
|
||||
self.approve = approve
|
||||
self.requests: List[Dict[str, Any]] = []
|
||||
|
||||
def request(self, action: Dict[str, Any]) -> bool:
|
||||
self.requests.append(action)
|
||||
return self.approve
|
||||
|
||||
|
||||
def _normalise(events: List[Dict[str, Any]], out_dir: Path) -> List[Dict[str, Any]]:
|
||||
"""Replace the run's own output path with a placeholder.
|
||||
|
||||
The two runs write into different temp folders, so absolute paths in
|
||||
``tool_result``/``outputs_*`` events differ by construction. Everything else
|
||||
must match verbatim.
|
||||
"""
|
||||
marker, raw = "<OUT>", str(out_dir)
|
||||
|
||||
def scrub(value: Any) -> Any:
|
||||
if isinstance(value, str):
|
||||
return value.replace(raw, marker).replace(raw.replace("\\", "/"), marker)
|
||||
if isinstance(value, list):
|
||||
return [scrub(v) for v in value]
|
||||
if isinstance(value, dict):
|
||||
return {k: scrub(v) for k, v in value.items()}
|
||||
return value
|
||||
|
||||
return [scrub(e) for e in events]
|
||||
|
||||
|
||||
def _run_legacy(tmp_path: Path, provider: FakeProvider, *, allowed_tools=None,
|
||||
gate: Optional[_FakeGate] = None, max_steps: int = 30
|
||||
) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]], List[str]]:
|
||||
"""Run the turn through the existing ``run_cowork``."""
|
||||
out_dir = tmp_path / "legacy"
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
events: List[Dict[str, Any]] = []
|
||||
messages = [dict(m) for m in _USER_TURN]
|
||||
|
||||
chat_agent.run_cowork(
|
||||
provider, messages, out_dir, events.append, title="Report",
|
||||
security_config=None, allowed_tools=allowed_tools, gate=gate, max_steps=max_steps,
|
||||
)
|
||||
tool_names = [t.name for t in (provider.last_tools or [])]
|
||||
return _normalise(events, out_dir), messages, tool_names
|
||||
|
||||
|
||||
def _run_service(tmp_path: Path, provider: FakeProvider, *, allowed_tools=None,
|
||||
gate: Optional[_FakeGate] = None, max_steps: int = 30
|
||||
) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]], List[str]]:
|
||||
"""Run the same turn through the application service."""
|
||||
out_dir = tmp_path / "service"
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
events: List[Dict[str, Any]] = []
|
||||
|
||||
service = build_cowork_conversation_service(
|
||||
provider, out_dir, events.append, title="Report", security_config=None, gate=gate)
|
||||
request = ConversationExecutionRequest(
|
||||
turn_id="t1", session_id="s1",
|
||||
# run_cowork receives the user message already appended; the request
|
||||
# carries the history and this turn's prompt separately.
|
||||
messages=_USER_TURN[:-1], prompt=_USER_TURN[-1]["content"],
|
||||
output_dir=out_dir, allowed_tools=allowed_tools, max_steps=max_steps,
|
||||
gate_mode="confirm" if gate is not None else "auto",
|
||||
)
|
||||
result = service.execute(request, legacy_event_sink(events.append))
|
||||
|
||||
# The end-of-turn event is new in R04 and has no legacy counterpart.
|
||||
kept = [e for e in events if e.get("type") != "turn_completed"]
|
||||
tool_names = [t.name for t in (provider.last_tools or [])]
|
||||
return _normalise(kept, out_dir), list(result.messages), tool_names
|
||||
|
||||
|
||||
def _assert_parity(tmp_path: Path, script, *, approve: Optional[bool] = None, **kwargs) -> None:
|
||||
"""Script two identical providers, run both paths, compare everything."""
|
||||
legacy_provider, service_provider = FakeProvider(), FakeProvider()
|
||||
script(legacy_provider)
|
||||
script(service_provider)
|
||||
|
||||
legacy_gate = _FakeGate(approve) if approve is not None else None
|
||||
service_gate = _FakeGate(approve) if approve is not None else None
|
||||
|
||||
legacy_events, legacy_messages, legacy_tools = _run_legacy(
|
||||
tmp_path, legacy_provider, gate=legacy_gate, **kwargs)
|
||||
service_events, service_messages, service_tools = _run_service(
|
||||
tmp_path, service_provider, gate=service_gate, **kwargs)
|
||||
|
||||
assert service_events == legacy_events
|
||||
assert service_messages == legacy_messages
|
||||
assert service_tools == legacy_tools
|
||||
if legacy_gate is not None and service_gate is not None:
|
||||
assert [r["name"] for r in service_gate.requests] == \
|
||||
[r["name"] for r in legacy_gate.requests]
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Scenarios.
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_a_plain_answer_turn_behaves_identically(tmp_path: Path) -> None:
|
||||
def script(provider: FakeProvider) -> None:
|
||||
provider.queue_response(content="Here you go.", chunks=["Here ", "you go."])
|
||||
|
||||
_assert_parity(tmp_path, script)
|
||||
|
||||
|
||||
def test_a_save_file_turn_behaves_identically(tmp_path: Path) -> None:
|
||||
def script(provider: FakeProvider) -> None:
|
||||
provider.queue_response(
|
||||
content="Writing it.",
|
||||
tool_calls=[{"id": "c1", "name": "save_file",
|
||||
"arguments": {"filename": "report.md", "content": "# Report\n"}}],
|
||||
)
|
||||
provider.queue_response(content="Saved.")
|
||||
|
||||
_assert_parity(tmp_path, script)
|
||||
|
||||
|
||||
def test_an_update_plan_turn_behaves_identically(tmp_path: Path) -> None:
|
||||
def script(provider: FakeProvider) -> None:
|
||||
provider.queue_response(
|
||||
content="Planning.",
|
||||
tool_calls=[{"id": "c1", "name": "update_plan",
|
||||
"arguments": {"steps": [{"title": "Draft", "status": "running"},
|
||||
{"title": "Ship", "status": "pending"}]}}],
|
||||
)
|
||||
provider.queue_response(content="Done.")
|
||||
|
||||
_assert_parity(tmp_path, script)
|
||||
|
||||
|
||||
def test_a_reasoning_only_reply_behaves_identically(tmp_path: Path) -> None:
|
||||
def script(provider: FakeProvider) -> None:
|
||||
provider.queue_response(content="", reasoning="thinking hard")
|
||||
|
||||
_assert_parity(tmp_path, script)
|
||||
|
||||
|
||||
def test_restricting_the_tool_scope_advertises_the_same_tools(tmp_path: Path) -> None:
|
||||
def script(provider: FakeProvider) -> None:
|
||||
provider.queue_response(content="ok")
|
||||
|
||||
_assert_parity(tmp_path, script, allowed_tools=["save_file"])
|
||||
|
||||
|
||||
def test_a_rejected_command_behaves_identically(tmp_path: Path) -> None:
|
||||
# The security-critical path: the gate says no, so the command must never
|
||||
# run and the model must read back the same refusal in both designs.
|
||||
def script(provider: FakeProvider) -> None:
|
||||
provider.queue_response(
|
||||
content="Running it.",
|
||||
tool_calls=[{"id": "c1", "name": "run_command",
|
||||
"arguments": {"command": "echo hi"}}],
|
||||
)
|
||||
provider.queue_response(content="Understood.")
|
||||
|
||||
_assert_parity(tmp_path, script, approve=False)
|
||||
|
||||
|
||||
def test_hitting_the_step_ceiling_behaves_identically(tmp_path: Path) -> None:
|
||||
# The model never stops calling tools, so both paths must stop at the same
|
||||
# place and say so the same way.
|
||||
def script(provider: FakeProvider) -> None:
|
||||
for i in range(4):
|
||||
provider.queue_response(
|
||||
content=f"step {i}",
|
||||
tool_calls=[{"id": f"c{i}", "name": "save_file",
|
||||
"arguments": {"filename": f"f{i}.md", "content": "x"}}],
|
||||
)
|
||||
|
||||
_assert_parity(tmp_path, script, max_steps=2)
|
||||
@@ -0,0 +1,220 @@
|
||||
"""R04-T04 — the migrated Cowork call site, exercised end to end without Qt.
|
||||
|
||||
``CoworkTab.build_job`` only ever *reads attributes* off its widget, so the real
|
||||
production method can be invoked against a stand-in that supplies those
|
||||
attributes. That is what happens here: the actual ``build_job`` body runs, builds
|
||||
a request, wires the service through ``core_runtime_adapter``, and drives a real
|
||||
turn (real tool execution, real output-folder cleanup) against ``FakeProvider``.
|
||||
|
||||
Why it matters: this is the only automated check that the widget's contract with
|
||||
the service still holds — that the worker's list is appended to in place (the
|
||||
transcript re-render and history merge both read it), that events still arrive as
|
||||
legacy dicts, and that a produced file really lands in the turn's folder. None of
|
||||
it needs a display server, so it runs in CI like every other test.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
import pytest
|
||||
from cowork_local.config import DEFAULT_CONFIG, AppConfig
|
||||
from cowork_local.tests.fakes.fake_provider import FakeProvider
|
||||
|
||||
|
||||
class _FakeWorker:
|
||||
"""The parts of ``core/worker.py::AgentWorker`` a job actually touches."""
|
||||
|
||||
def __init__(self, approve_commands: bool = True) -> None:
|
||||
self.events: List[Dict[str, Any]] = []
|
||||
self.gate: Optional[Any] = None
|
||||
self._approve = approve_commands
|
||||
self.cancelled = False
|
||||
|
||||
def emit_event(self, event: Dict[str, Any]) -> None:
|
||||
self.events.append(event)
|
||||
|
||||
def is_cancelled(self) -> bool:
|
||||
return self.cancelled
|
||||
|
||||
def new_gate(self, mode: str, agent_role: str = "") -> Any:
|
||||
# Mirrors AgentWorker.new_gate: the gate is stored on the worker so the
|
||||
# UI thread can resolve it, and answers request() from the worker thread.
|
||||
worker = self
|
||||
|
||||
class _Gate:
|
||||
requests: List[Dict[str, Any]] = []
|
||||
|
||||
def request(self, action: Dict[str, Any]) -> bool:
|
||||
self.requests.append(action)
|
||||
return worker._approve
|
||||
|
||||
self.gate = _Gate()
|
||||
return self.gate
|
||||
|
||||
|
||||
class _FakeCtx:
|
||||
"""The ``AppContext`` surface ``build_job`` uses."""
|
||||
|
||||
def __init__(self, config: AppConfig, confirm_commands: bool = False) -> None:
|
||||
self.config = config
|
||||
self._confirm = confirm_commands
|
||||
|
||||
def project_confirm_commands(self) -> bool:
|
||||
return self._confirm
|
||||
|
||||
def build_mcp_tools(self):
|
||||
return [], None
|
||||
|
||||
|
||||
class _WidgetStub:
|
||||
"""Stands in for the CoworkTab instance ``build_job`` reads its state from."""
|
||||
|
||||
kind = "cowork"
|
||||
|
||||
def __init__(self, out_root: Path, ctx: _FakeCtx, provider: FakeProvider) -> None:
|
||||
self._out_root = out_root
|
||||
self.ctx = ctx
|
||||
self._provider = provider
|
||||
self.title = "Report"
|
||||
self.session_id = "s1"
|
||||
self.project_id = "" # the auto-seeded default workspace
|
||||
self._model = ""
|
||||
self._routed_provider = None
|
||||
self._routed_model = None
|
||||
|
||||
def _session_output_dir(self) -> Path:
|
||||
return self._out_root
|
||||
|
||||
def workspace_dir(self) -> Path:
|
||||
return self._out_root
|
||||
|
||||
def admin_agent_prompt(self) -> str:
|
||||
return ""
|
||||
|
||||
def build_provider(self) -> FakeProvider:
|
||||
return self._provider
|
||||
|
||||
|
||||
def _config() -> AppConfig:
|
||||
"""A real AppConfig that never touches ``~/.cowork_local``.
|
||||
|
||||
The AI security guardrails are switched off: they would call the model to
|
||||
review the prompt, which is a separate feature with its own tests and would
|
||||
make this one depend on what the fake answers.
|
||||
"""
|
||||
data = copy.deepcopy(DEFAULT_CONFIG)
|
||||
data["agent_security"]["enabled"] = False
|
||||
return AppConfig(data)
|
||||
|
||||
|
||||
def _run_turn(tmp_path: Path, provider: FakeProvider, messages: List[Dict[str, Any]],
|
||||
*, confirm_commands: bool = False, approve: bool = True):
|
||||
"""Invoke the real ``CoworkTab.build_job`` against the stub and run its job."""
|
||||
from cowork_local.ui.cowork_tab import CoworkTab
|
||||
|
||||
out_dir = tmp_path / ".turns" / "t1"
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
widget = _WidgetStub(tmp_path, _FakeCtx(_config(), confirm_commands), provider)
|
||||
worker = _FakeWorker(approve_commands=approve)
|
||||
|
||||
job = CoworkTab.build_job(widget, "make me a report", messages, out_dir)
|
||||
result = job(worker)
|
||||
return result, worker
|
||||
|
||||
|
||||
def test_the_turn_runs_and_reports_its_folder(tmp_path: Path) -> None:
|
||||
provider = FakeProvider()
|
||||
provider.queue_response(content="Here you go.", chunks=["Here ", "you go."])
|
||||
messages = [{"role": "user", "content": "make me a report"}]
|
||||
|
||||
result, worker = _run_turn(tmp_path, provider, messages)
|
||||
|
||||
assert result["turn_dir"] == str(tmp_path / ".turns" / "t1")
|
||||
assert [e["type"] for e in worker.events] == [
|
||||
"text", "text", "assistant_done", "turn_completed"]
|
||||
|
||||
|
||||
def test_the_worker_list_is_appended_to_in_place(tmp_path: Path) -> None:
|
||||
# _reattach_running_turn replays from this very list while the turn runs, and
|
||||
# _finalize_turn slices it by the pre-turn length afterwards.
|
||||
provider = FakeProvider()
|
||||
provider.queue_response(content="Done.")
|
||||
user = {"role": "user", "content": "make me a report"}
|
||||
messages = [user]
|
||||
|
||||
result, _ = _run_turn(tmp_path, provider, messages)
|
||||
|
||||
assert result["messages"] is messages
|
||||
# Identity, not just equality: _reattach_running_turn locates the turn's user
|
||||
# message with ``m is ctx["user_msg"]`` to replay the steps after it.
|
||||
assert any(m is user for m in messages)
|
||||
# Several system blocks are expected — the tool prompt plus the tagged
|
||||
# skills/security-rules blocks the runtime refreshes on every turn.
|
||||
assert [m["role"] for m in messages if m["role"] != "system"] == ["user", "assistant"]
|
||||
assert messages[-1]["content"] == "Done."
|
||||
|
||||
|
||||
def test_a_saved_file_lands_in_the_turn_folder(tmp_path: Path) -> None:
|
||||
provider = FakeProvider()
|
||||
provider.queue_response(
|
||||
content="Writing it.",
|
||||
tool_calls=[{"id": "c1", "name": "save_file",
|
||||
"arguments": {"filename": "report.md", "content": "# Report\n"}}],
|
||||
)
|
||||
provider.queue_response(content="Saved.")
|
||||
messages = [{"role": "user", "content": "make me a report"}]
|
||||
|
||||
_, worker = _run_turn(tmp_path, provider, messages)
|
||||
|
||||
produced = list((tmp_path / ".turns" / "t1").glob("*.md"))
|
||||
assert len(produced) == 1
|
||||
assert produced[0].read_text(encoding="utf-8") == "# Report\n"
|
||||
results = [e for e in worker.events if e["type"] == "tool_result"]
|
||||
assert results and results[0]["ok"] is True
|
||||
|
||||
|
||||
def test_auto_run_mode_never_creates_a_permission_gate(tmp_path: Path) -> None:
|
||||
provider = FakeProvider()
|
||||
provider.queue_response(content="ok")
|
||||
|
||||
_, worker = _run_turn(tmp_path, provider, [{"role": "user", "content": "hi"}],
|
||||
confirm_commands=False)
|
||||
|
||||
assert worker.gate is None
|
||||
|
||||
|
||||
def test_confirm_mode_creates_the_gate_and_a_refusal_stops_the_command(tmp_path: Path) -> None:
|
||||
provider = FakeProvider()
|
||||
provider.queue_response(
|
||||
content="Running it.",
|
||||
tool_calls=[{"id": "c1", "name": "run_command",
|
||||
"arguments": {"command": "echo hi"}}],
|
||||
)
|
||||
provider.queue_response(content="Understood.")
|
||||
|
||||
_, worker = _run_turn(tmp_path, provider, [{"role": "user", "content": "run it"}],
|
||||
confirm_commands=True, approve=False)
|
||||
|
||||
assert worker.gate is not None
|
||||
refusals = [e for e in worker.events
|
||||
if e["type"] == "tool_result" and e["output"] == "Rejected by user."]
|
||||
assert len(refusals) == 1
|
||||
|
||||
|
||||
def test_cancelling_before_the_turn_starts_calls_no_model(tmp_path: Path) -> None:
|
||||
from cowork_local.ui.cowork_tab import CoworkTab
|
||||
|
||||
provider = FakeProvider()
|
||||
provider.queue_response(content="never")
|
||||
out_dir = tmp_path / ".turns" / "t1"
|
||||
out_dir.mkdir(parents=True)
|
||||
widget = _WidgetStub(tmp_path, _FakeCtx(_config()), provider)
|
||||
worker = _FakeWorker()
|
||||
worker.cancelled = True
|
||||
|
||||
CoworkTab.build_job(widget, "x", [{"role": "user", "content": "x"}], out_dir)(worker)
|
||||
|
||||
assert provider.call_count == 0
|
||||
@@ -0,0 +1,191 @@
|
||||
"""R04-T05 — the Schedule Task runner's cowork branch, pinned before and after.
|
||||
|
||||
Written against the CURRENT ``_run_agent`` first, as the safety net for moving it
|
||||
onto ``ConversationApplicationService``: an unattended run has five behaviours the
|
||||
interactive path does not have (the plan reminder prefixed to the prompt, the
|
||||
session registered in History before the model starts, a re-save after every
|
||||
assistant message, the timeout notice, and the "did the agent's own checklist
|
||||
finish?" report), and none of them was covered by a test.
|
||||
|
||||
Everything is isolated from the user's real config: history goes to ``tmp_path``
|
||||
via ``history.custom_dir`` and the AI guardrails are off, so no run touches
|
||||
``~/.cowork_local`` or calls a model to review a prompt.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from cowork_local.config import DEFAULT_CONFIG, AppConfig
|
||||
from cowork_local.core import task_executors
|
||||
from cowork_local.tests.fakes.fake_provider import FakeProvider
|
||||
|
||||
|
||||
class _FakeCtx:
|
||||
"""The ``AppContext`` surface ``_run_agent`` touches."""
|
||||
|
||||
def __init__(self, config: AppConfig, provider: FakeProvider) -> None:
|
||||
self.config = config
|
||||
self._provider = provider
|
||||
|
||||
def build_active_provider(self) -> FakeProvider:
|
||||
return self._provider
|
||||
|
||||
def build_provider_for(self, name=None, model=None) -> FakeProvider:
|
||||
return self._provider
|
||||
|
||||
|
||||
def _config(tmp_path: Path) -> AppConfig:
|
||||
data = copy.deepcopy(DEFAULT_CONFIG)
|
||||
# Keep the run entirely offline and off the real config dir.
|
||||
data["agent_security"]["enabled"] = False
|
||||
data["history"]["custom_dir"] = str(tmp_path / "history")
|
||||
return AppConfig(data)
|
||||
|
||||
|
||||
def _run(tmp_path: Path, provider: FakeProvider, *, prompt: str = "write the report",
|
||||
timeout_sec: Optional[int] = None, admin_agent: Any = None):
|
||||
"""Run one cowork task and return ``(result_tuple, events, config)``."""
|
||||
out_dir = tmp_path / "run"
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
config = _config(tmp_path)
|
||||
events: List[Dict[str, Any]] = []
|
||||
|
||||
result = task_executors._run_agent(
|
||||
_FakeCtx(config, provider), "cowork", prompt, out_dir,
|
||||
events.append, lambda: False, title="Weekly report",
|
||||
timeout_sec=timeout_sec, admin_agent=admin_agent,
|
||||
)
|
||||
return result, events, config
|
||||
|
||||
|
||||
def _saved_conversation(config: AppConfig) -> Dict[str, Any]:
|
||||
"""The single conversation the run wrote into the isolated history folder."""
|
||||
files = list(Path(config.history_dir()).rglob("*.json"))
|
||||
assert len(files) == 1, f"expected one saved conversation, found {files}"
|
||||
return json.loads(files[0].read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_a_cowork_task_returns_the_final_answer(tmp_path: Path) -> None:
|
||||
provider = FakeProvider()
|
||||
provider.queue_response(content="Report is ready.")
|
||||
|
||||
(answer, timed_out, incomplete), _events, _config = _run(tmp_path, provider)
|
||||
|
||||
assert answer == "Report is ready."
|
||||
assert timed_out is False
|
||||
assert incomplete == ""
|
||||
|
||||
|
||||
def test_the_plan_reminder_is_prefixed_to_the_prompt(tmp_path: Path) -> None:
|
||||
# An unattended run has nobody watching, so the agent is pushed to keep its
|
||||
# own checklist honest. The reminder must lead the message.
|
||||
provider = FakeProvider()
|
||||
provider.queue_response(content="ok")
|
||||
|
||||
_run(tmp_path, provider, prompt="write the report")
|
||||
|
||||
sent = provider.call_history[0][-1]["content"]
|
||||
assert sent.startswith("This runs unattended (Schedule Task)")
|
||||
assert sent.endswith("write the report")
|
||||
|
||||
|
||||
def test_an_admin_agent_persona_sits_between_the_reminder_and_the_prompt(
|
||||
tmp_path: Path) -> None:
|
||||
class _Agent:
|
||||
# An admin agent may pin its own provider/model; blank means "use the
|
||||
# machine's Settings default", which is what build_agent_provider reads.
|
||||
provider = ""
|
||||
model = ""
|
||||
|
||||
def effective_prompt(self) -> str:
|
||||
return "You are the reporting agent."
|
||||
|
||||
provider = FakeProvider()
|
||||
provider.queue_response(content="ok")
|
||||
|
||||
_run(tmp_path, provider, prompt="write the report", admin_agent=_Agent())
|
||||
|
||||
sent = provider.call_history[0][-1]["content"]
|
||||
assert sent.index("This runs unattended") < sent.index("You are the reporting agent.")
|
||||
assert sent.index("You are the reporting agent.") < sent.index("write the report")
|
||||
|
||||
|
||||
def test_the_session_is_announced_once_it_exists_on_disk(tmp_path: Path) -> None:
|
||||
# The scheduler refreshes History on this event, so it must not fire before
|
||||
# the conversation is really there.
|
||||
provider = FakeProvider()
|
||||
provider.queue_response(content="ok")
|
||||
|
||||
_result, events, config = _run(tmp_path, provider)
|
||||
|
||||
ready = [e for e in events if e["type"] == "history_ready"]
|
||||
assert len(ready) == 1
|
||||
assert ready[0]["session_id"]
|
||||
assert _saved_conversation(config)["session_id"] == ready[0]["session_id"]
|
||||
|
||||
|
||||
def test_the_saved_conversation_carries_the_answer_and_the_task_title(
|
||||
tmp_path: Path) -> None:
|
||||
provider = FakeProvider()
|
||||
provider.queue_response(content="Report is ready.")
|
||||
|
||||
_result, _events, config = _run(tmp_path, provider)
|
||||
|
||||
saved = _saved_conversation(config)
|
||||
assert saved["title"] == "[Task] Weekly report"
|
||||
assert saved["messages"][-1] == {"role": "assistant", "content": "Report is ready."}
|
||||
|
||||
|
||||
def test_an_unfinished_checklist_is_reported_back_to_the_scheduler(
|
||||
tmp_path: Path) -> None:
|
||||
# The agent ticked no step to done, so the task must not be called finished
|
||||
# just because no exception was raised.
|
||||
provider = FakeProvider()
|
||||
provider.queue_response(
|
||||
content="Working on it.",
|
||||
tool_calls=[{"id": "c1", "name": "update_plan",
|
||||
"arguments": {"steps": [{"title": "Draft", "status": "running"}]}}],
|
||||
)
|
||||
provider.queue_response(content="Stopping here.")
|
||||
|
||||
(_answer, _timed_out, incomplete), _events, _config = _run(tmp_path, provider)
|
||||
|
||||
assert incomplete
|
||||
assert "Draft" in incomplete
|
||||
|
||||
|
||||
def test_a_finished_checklist_reports_nothing_outstanding(tmp_path: Path) -> None:
|
||||
provider = FakeProvider()
|
||||
provider.queue_response(
|
||||
content="Done.",
|
||||
tool_calls=[{"id": "c1", "name": "update_plan",
|
||||
"arguments": {"steps": [{"title": "Draft", "status": "done"}]}}],
|
||||
)
|
||||
provider.queue_response(content="All done.")
|
||||
|
||||
(_answer, _timed_out, incomplete), _events, _config = _run(tmp_path, provider)
|
||||
|
||||
assert incomplete == ""
|
||||
|
||||
|
||||
def test_running_out_of_time_appends_the_timeout_notice_to_the_conversation(
|
||||
tmp_path: Path) -> None:
|
||||
# A negative timeout puts the deadline in the past, which is the only
|
||||
# deterministic way to exercise a wall-clock branch in a unit test.
|
||||
provider = FakeProvider()
|
||||
provider.queue_response(content="never gets there")
|
||||
|
||||
(answer, timed_out, incomplete), events, config = _run(
|
||||
tmp_path, provider, timeout_sec=-1)
|
||||
|
||||
assert timed_out is True
|
||||
assert incomplete == "" # a timeout is not an unfinished checklist
|
||||
assert "quá thời gian chờ" in answer
|
||||
assert any(e["type"] == "assistant_done" and "quá thời gian chờ" in e["content"]
|
||||
for e in events)
|
||||
assert "quá thời gian chờ" in _saved_conversation(config)["messages"][-1]["content"]
|
||||
@@ -0,0 +1,293 @@
|
||||
"""R04-T03 (b) — the turn loop: composition, tool dispatch, budget, cancel.
|
||||
|
||||
Behaviour that used to be reachable only by running the real widget. Every
|
||||
dependency is a fake from ``tests/fakes/turn_runtime_fakes.py``, so the file
|
||||
runs in milliseconds and each test states one rule of the loop.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, List, Tuple
|
||||
|
||||
from cowork_local.application.conversations.conversation_application_service import (
|
||||
ConversationApplicationService,
|
||||
)
|
||||
from cowork_local.domain.agents.agent_event import (
|
||||
AssistantMessageCompletedEvent,
|
||||
PlanStep,
|
||||
PlanUpdatedEvent,
|
||||
TextChunkEvent,
|
||||
ToolCallFinishedEvent,
|
||||
ToolCallStartedEvent,
|
||||
ToolOutputChunkEvent,
|
||||
ToolPreview,
|
||||
TurnCompletedEvent,
|
||||
)
|
||||
from cowork_local.tests.fakes.turn_runtime_fakes import (
|
||||
FakeModelCall,
|
||||
FakeReply,
|
||||
FakeToolRuntime,
|
||||
events_of_type,
|
||||
make_request,
|
||||
run_turn,
|
||||
tool_turn,
|
||||
)
|
||||
|
||||
|
||||
def _service(model, tools, **overrides) -> ConversationApplicationService:
|
||||
return ConversationApplicationService(model, tools, **overrides)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# The happy path.
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_a_plain_answer_streams_text_then_reports_the_message_and_the_turn() -> None:
|
||||
model = FakeModelCall([FakeReply(content="Hello there", chunks=["Hello ", "there"])])
|
||||
|
||||
result, events = run_turn(_service(model, FakeToolRuntime()))
|
||||
|
||||
assert [e.delta for e in events_of_type(events, TextChunkEvent)] == ["Hello ", "there"]
|
||||
assert events_of_type(events, AssistantMessageCompletedEvent) == [
|
||||
AssistantMessageCompletedEvent(content="Hello there")]
|
||||
assert events_of_type(events, TurnCompletedEvent) == [
|
||||
TurnCompletedEvent(final_text="Hello there", steps_used=1)]
|
||||
assert result.final_text == "Hello there"
|
||||
assert result.ok is True
|
||||
|
||||
|
||||
def test_the_composed_user_message_is_appended_before_the_first_call() -> None:
|
||||
model = FakeModelCall([FakeReply(content="ok")])
|
||||
request = make_request(prompt="ship it", instruction_prefix="RULES",
|
||||
session_notes="earlier: a.md",
|
||||
messages=[{"role": "user", "content": "previous"}])
|
||||
|
||||
run_turn(_service(model, FakeToolRuntime()), request)
|
||||
|
||||
sent = model.calls[0]["messages"]
|
||||
assert sent[-1] == {"role": "user",
|
||||
"content": "RULES\n\n---\n\nship it\n\nearlier: a.md"}
|
||||
assert sent[-2] == {"role": "user", "content": "previous"}
|
||||
|
||||
|
||||
def test_attachments_are_read_when_the_turn_runs_not_when_it_was_built() -> None:
|
||||
# Extraction can pip-install a parser or shell out to LibreOffice, so it must
|
||||
# happen here (worker thread), not while the UI was assembling the request.
|
||||
seen: List[Tuple[str, Tuple[str, ...]]] = []
|
||||
|
||||
def reader(prompt: str, attachments: Tuple[str, ...]) -> str:
|
||||
seen.append((prompt, attachments))
|
||||
return f"{prompt}\n\n<contents of {len(attachments)} file(s)>"
|
||||
|
||||
model = FakeModelCall([FakeReply(content="ok")])
|
||||
request = make_request(prompt="summarise", attachments=["a.docx", "b.pdf"])
|
||||
|
||||
run_turn(_service(model, FakeToolRuntime(), attachment_reader=reader), request)
|
||||
|
||||
assert seen == [("summarise", ("a.docx", "b.pdf"))]
|
||||
assert "contents of 2 file(s)" in model.calls[0]["messages"][-1]["content"]
|
||||
|
||||
|
||||
def test_the_prompt_preparer_is_told_which_tools_the_turn_advertises() -> None:
|
||||
# The system prompt gains an MS365 paragraph only when ms365__* tools are
|
||||
# present, so the preparer has to see the real list.
|
||||
seen: List[Tuple[str, ...]] = []
|
||||
model = FakeModelCall([FakeReply(content="ok")])
|
||||
tools = FakeToolRuntime(specs=("save_file", "ms365__send_mail"))
|
||||
|
||||
run_turn(_service(model, tools,
|
||||
prepare_prompt=lambda messages, names: seen.append(names)))
|
||||
|
||||
assert seen == [("save_file", "ms365__send_mail")]
|
||||
|
||||
|
||||
def test_only_the_allowed_tools_are_advertised() -> None:
|
||||
model = FakeModelCall([FakeReply(content="ok")])
|
||||
tools = FakeToolRuntime(specs=("save_file", "run_command", "update_plan"))
|
||||
|
||||
run_turn(_service(model, tools), make_request(allowed_tools=("save_file", "update_plan")))
|
||||
|
||||
assert model.calls[0]["tool_names"] == ["save_file", "update_plan"]
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Tool dispatch.
|
||||
# --------------------------------------------------------------------------- #
|
||||
def tool_turn(tool_name: str = "save_file", args=None, **tool_kwargs):
|
||||
"""A turn that calls one tool, then answers."""
|
||||
calls = [{"id": "c1", "name": tool_name, "arguments": args or {"filename": "a.md"}}]
|
||||
model = FakeModelCall([FakeReply(content="working", tool_calls=calls),
|
||||
FakeReply(content="done")])
|
||||
return model, FakeToolRuntime(**tool_kwargs)
|
||||
|
||||
|
||||
def test_a_tool_call_is_announced_executed_and_answered_in_the_message_list() -> None:
|
||||
model, tools = tool_turn(results={"save_file": {"ok": True, "output": "saved",
|
||||
"path": "out/a.md"}})
|
||||
|
||||
result, events = run_turn(_service(model, tools))
|
||||
|
||||
assert events_of_type(events, ToolCallStartedEvent) == [ToolCallStartedEvent(
|
||||
call_id="c1", name="save_file", arguments={"filename": "a.md"},
|
||||
preview=ToolPreview(kind="info", title="save_file", text="{'filename': 'a.md'}"))]
|
||||
assert events_of_type(events, ToolCallFinishedEvent) == [ToolCallFinishedEvent(
|
||||
call_id="c1", name="save_file", ok=True, output="saved", path="out/a.md")]
|
||||
assert tools.executed == [("save_file", {"filename": "a.md"})]
|
||||
assert result.messages[-2] == {"role": "tool", "tool_call_id": "c1",
|
||||
"name": "save_file", "content": "saved"}
|
||||
|
||||
|
||||
def test_live_tool_output_is_streamed_while_the_tool_runs() -> None:
|
||||
model, tools = tool_turn("run_command", {"command": "ls"})
|
||||
tools.emit_output = "file-a\n"
|
||||
|
||||
_, events = run_turn(_service(model, tools))
|
||||
|
||||
assert events_of_type(events, ToolOutputChunkEvent) == [ToolOutputChunkEvent(
|
||||
call_id="c1", name="run_command", delta="file-a\n")]
|
||||
|
||||
|
||||
def test_the_loop_ends_as_soon_as_the_model_stops_calling_tools() -> None:
|
||||
model, tools = tool_turn()
|
||||
|
||||
result, _ = run_turn(_service(model, tools))
|
||||
|
||||
assert result.steps_used == 2
|
||||
assert result.budget_exhausted is False
|
||||
|
||||
|
||||
def test_the_plan_tool_reports_a_plan_update_and_no_tool_bubble() -> None:
|
||||
calls = [{"id": "c1", "name": "update_plan",
|
||||
"arguments": {"steps": [{"title": "Draft", "status": "running"}]}}]
|
||||
model = FakeModelCall([FakeReply(content="planning", tool_calls=calls), FakeReply(content="done")])
|
||||
tools = FakeToolRuntime(results={"update_plan": {
|
||||
"ok": True, "output": "Plan updated.",
|
||||
"plan_steps": [PlanStep(title="Draft", status="running")]}})
|
||||
|
||||
result, events = run_turn(_service(model, tools))
|
||||
|
||||
assert events_of_type(events, PlanUpdatedEvent) == [
|
||||
PlanUpdatedEvent(steps=(PlanStep(title="Draft", status="running"),))]
|
||||
assert events_of_type(events, ToolCallStartedEvent) == []
|
||||
assert events_of_type(events, ToolCallFinishedEvent) == []
|
||||
assert result.plan_steps == (PlanStep(title="Draft", status="running"),)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Budget, cancellation.
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_running_out_of_steps_is_flagged_and_announced() -> None:
|
||||
# The model keeps calling tools forever; the ceiling must stop it visibly.
|
||||
forever = [FakeReply(content=f"step {i}",
|
||||
tool_calls=[{"id": f"c{i}", "name": "save_file", "arguments": {}}])
|
||||
for i in range(5)]
|
||||
model = FakeModelCall(forever)
|
||||
|
||||
result, events = run_turn(_service(model, FakeToolRuntime()), make_request(max_steps=2))
|
||||
|
||||
assert result.steps_used == 2
|
||||
assert result.budget_exhausted is True
|
||||
assert "2-step safety limit" in events_of_type(events, TextChunkEvent)[-1].delta
|
||||
# The note reaches the transcript but NOT the stored answer: a turn that hits
|
||||
# the ceiling always ends on a tool message, and the existing runtime only
|
||||
# merges the note when the last message is the assistant's. Pinned here so a
|
||||
# future change to that rule is a deliberate decision, not a silent drift.
|
||||
assert result.final_text == "step 1"
|
||||
|
||||
|
||||
def test_run_to_completion_uses_the_higher_ceiling() -> None:
|
||||
forever = [FakeReply(content="x", tool_calls=[{"id": "c", "name": "save_file", "arguments": {}}])
|
||||
for _ in range(6)]
|
||||
model = FakeModelCall(forever)
|
||||
|
||||
result, _ = run_turn(_service(model, FakeToolRuntime()),
|
||||
make_request(max_steps=2, completion_max_steps=5, run_to_completion=True))
|
||||
|
||||
assert result.steps_used == 5
|
||||
|
||||
|
||||
def test_a_turn_cancelled_before_it_starts_never_calls_the_model() -> None:
|
||||
model = FakeModelCall([FakeReply(content="never")])
|
||||
|
||||
result, events = run_turn(_service(model, FakeToolRuntime()), cancel=lambda: True)
|
||||
|
||||
assert model.calls == []
|
||||
assert result.cancelled is True
|
||||
assert result.budget_exhausted is False
|
||||
assert events_of_type(events, TurnCompletedEvent) == [TurnCompletedEvent(cancelled=True)]
|
||||
|
||||
|
||||
def test_cancelling_during_a_turn_stops_dispatching_the_remaining_tool_calls() -> None:
|
||||
calls = [{"id": "c1", "name": "save_file", "arguments": {}},
|
||||
{"id": "c2", "name": "save_file", "arguments": {}}]
|
||||
model = FakeModelCall([FakeReply(content="two tools", tool_calls=calls)])
|
||||
tools = FakeToolRuntime()
|
||||
stop = {"now": False}
|
||||
|
||||
def cancel() -> bool:
|
||||
return stop["now"]
|
||||
|
||||
original_execute = tools.execute
|
||||
|
||||
def execute(name, args, on_output=None, cancel=None):
|
||||
stop["now"] = True # cancel raised while the first tool runs
|
||||
return original_execute(name, args, on_output=on_output, cancel=cancel)
|
||||
|
||||
tools.execute = execute
|
||||
|
||||
result, _ = run_turn(_service(model, tools), cancel=cancel)
|
||||
|
||||
assert len(tools.executed) == 1
|
||||
assert result.cancelled is True
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Bring-your-own working list.
|
||||
#
|
||||
# ``ui/chat_panel.py`` holds the turn's message list in its own turn context and
|
||||
# reads it WHILE the worker appends (``_reattach_running_turn`` replays the steps
|
||||
# done so far when the user reopens a running conversation; ``_finalize_turn``
|
||||
# slices it by ``snapshot_len``). A service that built its own private list would
|
||||
# silently break both, so a caller can hand its list over instead.
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_a_caller_supplied_list_is_appended_to_in_place() -> None:
|
||||
model, tools = tool_turn()
|
||||
live: List[Dict[str, Any]] = [{"role": "user", "content": "already composed"}]
|
||||
|
||||
result = ConversationApplicationService(model, tools).execute(
|
||||
make_request(), lambda event: None, messages=live)
|
||||
|
||||
roles = [m["role"] for m in live]
|
||||
assert roles == ["user", "assistant", "tool", "assistant"]
|
||||
assert result.messages == tuple(live)
|
||||
|
||||
|
||||
def test_a_caller_supplied_list_is_used_as_is_without_recomposing_the_prompt() -> None:
|
||||
# The widget already applied the skill prefix and the session notes when it
|
||||
# built its message; composing again would duplicate them.
|
||||
model = FakeModelCall([FakeReply(content="ok")])
|
||||
user = {"role": "user", "content": "already composed"}
|
||||
live = [user]
|
||||
|
||||
ConversationApplicationService(model, FakeToolRuntime()).execute(
|
||||
make_request(prompt="typed text", instruction_prefix="RULES",
|
||||
session_notes="notes"),
|
||||
lambda event: None, messages=live)
|
||||
|
||||
assert live[0] is user
|
||||
assert live[0]["content"] == "already composed"
|
||||
assert [m["role"] for m in live].count("user") == 1
|
||||
|
||||
|
||||
def test_a_caller_supplied_list_skips_the_attachment_reader() -> None:
|
||||
# Reading the attachments is what produced the caller's message in the first
|
||||
# place; doing it again would re-parse every file.
|
||||
model = FakeModelCall([FakeReply(content="ok")])
|
||||
calls: List[Any] = []
|
||||
|
||||
ConversationApplicationService(
|
||||
model, FakeToolRuntime(),
|
||||
attachment_reader=lambda prompt, attachments: calls.append(prompt) or prompt,
|
||||
).execute(make_request(attachments=["a.docx"]), lambda event: None,
|
||||
messages=[{"role": "user", "content": "composed"}])
|
||||
|
||||
assert calls == []
|
||||
@@ -0,0 +1,210 @@
|
||||
"""R04-T03 (b) — the turn loop: guards, permission gate, compaction, cleanup.
|
||||
|
||||
Split out of ``test_conversation_application_service.py`` to keep each file
|
||||
inside the 400-LOC limit. Same fakes, same service; this half pins the ORDER of
|
||||
the safety steps (guard before model, guard before execute, gate before execute)
|
||||
and the promise that the output sandbox is tidied on the way out.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, List
|
||||
|
||||
import pytest
|
||||
from cowork_local.application.conversations.conversation_application_service import (
|
||||
ConversationApplicationService,
|
||||
)
|
||||
from cowork_local.domain.agents.agent_event import (
|
||||
ErrorEvent,
|
||||
OutputsAddedEvent,
|
||||
ReasoningChunkEvent,
|
||||
TextChunkEvent,
|
||||
ToolCallFinishedEvent,
|
||||
)
|
||||
from cowork_local.tests.fakes.turn_runtime_fakes import (
|
||||
FakeModelCall,
|
||||
FakeReply,
|
||||
FakeToolRuntime,
|
||||
events_of_type,
|
||||
make_request,
|
||||
run_turn,
|
||||
tool_turn,
|
||||
)
|
||||
|
||||
|
||||
def _service(model, tools, **overrides) -> ConversationApplicationService:
|
||||
return ConversationApplicationService(model, tools, **overrides)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Guards and the permission gate.
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_the_prompt_guard_runs_before_the_model_is_ever_called() -> None:
|
||||
order: List[str] = []
|
||||
model = FakeModelCall([FakeReply(content="ok")])
|
||||
model_call = model.call
|
||||
|
||||
def call(*a, **kw):
|
||||
order.append("model")
|
||||
return model_call(*a, **kw)
|
||||
|
||||
model.call = call
|
||||
|
||||
run_turn(_service(model, FakeToolRuntime(), prompt_guard=lambda messages: order.append("guard")))
|
||||
|
||||
assert order == ["guard", "model"]
|
||||
|
||||
|
||||
def test_a_blocked_prompt_propagates_before_the_output_folder_is_touched() -> None:
|
||||
model = FakeModelCall([FakeReply(content="never")])
|
||||
tools = FakeToolRuntime()
|
||||
events: List[Any] = []
|
||||
|
||||
def guard(messages) -> None:
|
||||
raise RuntimeError("SecurityBlocked: nope")
|
||||
|
||||
service = _service(model, tools, prompt_guard=guard)
|
||||
|
||||
with pytest.raises(RuntimeError, match="SecurityBlocked"):
|
||||
service.execute(make_request(), events.append)
|
||||
|
||||
assert model.calls == []
|
||||
assert events_of_type(events, ErrorEvent) == [ErrorEvent(message="SecurityBlocked: nope")]
|
||||
# Cleanup is NOT a read-only operation (it deletes a stale .scratch and every
|
||||
# empty sub-folder), so a turn rejected before it started must not run it.
|
||||
assert tools.finalize_calls == []
|
||||
|
||||
|
||||
def test_output_cleanup_still_runs_when_the_turn_fails_mid_loop() -> None:
|
||||
# Once the turn has started producing files, the sandbox must be tidied on
|
||||
# the way out no matter how the turn ends.
|
||||
model = FakeModelCall([RuntimeError("gateway exploded")])
|
||||
tools = FakeToolRuntime()
|
||||
events: List[Any] = []
|
||||
|
||||
with pytest.raises(RuntimeError, match="gateway exploded"):
|
||||
_service(model, tools).execute(make_request(), events.append)
|
||||
|
||||
assert tools.finalize_calls == [{"before": "before", "cancelled": False}]
|
||||
assert events_of_type(events, ErrorEvent) == [ErrorEvent(message="gateway exploded")]
|
||||
|
||||
|
||||
def test_the_command_guard_runs_before_the_tool_executes() -> None:
|
||||
order: List[str] = []
|
||||
model, tools = tool_turn("run_command", {"command": "ls"})
|
||||
original = tools.execute
|
||||
|
||||
def execute(name, args, on_output=None, cancel=None):
|
||||
order.append("execute")
|
||||
return original(name, args, on_output=on_output, cancel=cancel)
|
||||
|
||||
tools.execute = execute
|
||||
|
||||
run_turn(_service(model, tools,
|
||||
command_guard=lambda name, args: order.append(f"guard:{name}")))
|
||||
|
||||
assert order == ["guard:run_command", "execute"]
|
||||
|
||||
|
||||
def test_disabling_rule_enforcement_skips_both_guards() -> None:
|
||||
# Co4E flow steps run inside the workspace sandbox and opt out on purpose.
|
||||
calls: List[str] = []
|
||||
model, tools = tool_turn("run_command", {"command": "ls"})
|
||||
|
||||
run_turn(_service(model, tools,
|
||||
prompt_guard=lambda messages: calls.append("prompt"),
|
||||
command_guard=lambda name, args: calls.append("command")),
|
||||
make_request(enforce_rules=False))
|
||||
|
||||
assert calls == []
|
||||
|
||||
|
||||
def test_the_permission_gate_is_asked_only_for_command_tools() -> None:
|
||||
asked: List[str] = []
|
||||
model, tools = tool_turn("save_file", {"filename": "a.md"})
|
||||
|
||||
run_turn(_service(model, tools,
|
||||
permission_request=lambda action: asked.append(action["name"]) or True),
|
||||
make_request(gate_mode="confirm"))
|
||||
|
||||
assert asked == [] # save_file writes into the sandbox: never gated
|
||||
|
||||
|
||||
def test_a_command_tool_in_confirm_mode_asks_before_running() -> None:
|
||||
asked: List[Dict[str, Any]] = []
|
||||
model, tools = tool_turn("run_command", {"command": "ls"})
|
||||
|
||||
def approve(action: Dict[str, Any]) -> bool:
|
||||
asked.append(action)
|
||||
return True
|
||||
|
||||
run_turn(_service(model, tools, permission_request=approve), make_request(gate_mode="confirm"))
|
||||
|
||||
assert [a["name"] for a in asked] == ["run_command"]
|
||||
assert tools.executed == [("run_command", {"command": "ls"})]
|
||||
|
||||
|
||||
def test_a_rejected_command_is_reported_as_a_failed_tool_and_never_runs() -> None:
|
||||
model, tools = tool_turn("run_command", {"command": "rm -rf /"})
|
||||
|
||||
result, events = run_turn(_service(model, tools, permission_request=lambda action: False),
|
||||
make_request(gate_mode="confirm"))
|
||||
|
||||
assert tools.executed == []
|
||||
assert events_of_type(events, ToolCallFinishedEvent) == [ToolCallFinishedEvent(
|
||||
call_id="c1", name="run_command", ok=False, output="Rejected by user.")]
|
||||
assert result.messages[-2]["content"] == "Rejected by user."
|
||||
|
||||
|
||||
def test_auto_mode_never_asks_even_for_a_command() -> None:
|
||||
model, tools = tool_turn("run_command", {"command": "ls"})
|
||||
|
||||
def refuse(action): # would block the turn if it were consulted
|
||||
raise AssertionError("the gate must not be consulted in auto mode")
|
||||
|
||||
run_turn(_service(model, tools, permission_request=refuse), make_request(gate_mode="auto"))
|
||||
|
||||
assert tools.executed == [("run_command", {"command": "ls"})]
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Context compaction, reasoning, output cleanup.
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_the_conversation_is_offered_for_compaction_before_every_call() -> None:
|
||||
compactions: List[int] = []
|
||||
model, tools = tool_turn()
|
||||
|
||||
run_turn(_service(model, tools,
|
||||
compact=lambda messages, cancel: compactions.append(len(messages))))
|
||||
|
||||
assert len(compactions) == 2 # once per provider call
|
||||
|
||||
|
||||
def test_reasoning_is_streamed_as_its_own_event() -> None:
|
||||
model = FakeModelCall([FakeReply(content="42", reasoning="thinking...")])
|
||||
|
||||
_, events = run_turn(_service(model, FakeToolRuntime()))
|
||||
|
||||
assert events_of_type(events, ReasoningChunkEvent) == [ReasoningChunkEvent(delta="thinking...")]
|
||||
|
||||
|
||||
def test_a_reasoning_only_reply_gets_a_visible_note_in_the_transcript() -> None:
|
||||
# Otherwise a Schedule Task run reads back an empty answer and writes
|
||||
# "(no output)" into its report.
|
||||
model = FakeModelCall([FakeReply(content="", reasoning="thought hard")])
|
||||
|
||||
result, events = run_turn(_service(model, FakeToolRuntime()))
|
||||
|
||||
assert "only its reasoning" in events_of_type(events, TextChunkEvent)[-1].delta
|
||||
assert "only its reasoning" in result.final_text
|
||||
|
||||
|
||||
def test_promoted_and_discarded_output_files_are_reported_at_the_end() -> None:
|
||||
model = FakeModelCall([FakeReply(content="ok")])
|
||||
tools = FakeToolRuntime(added=("out/report.pptx",))
|
||||
|
||||
_, events = run_turn(_service(model, tools))
|
||||
|
||||
assert events_of_type(events, OutputsAddedEvent) == [
|
||||
OutputsAddedEvent(paths=("out/report.pptx",))]
|
||||
assert tools.finalize_calls == [{"before": "before", "cancelled": False}]
|
||||
@@ -0,0 +1,76 @@
|
||||
"""R04-T04 — unit tests for the UI-state -> request mapping.
|
||||
|
||||
Three small rules used to sit inline in ``ui/cowork_tab.py::build_job``, where no
|
||||
test could reach them: the turn's prompt is the last message in the working list,
|
||||
the history is everything before it, and the confirm-commands flag becomes a gate
|
||||
mode. Getting any of them wrong is silent (a duplicated user message, a command
|
||||
that stops asking for approval), so they are pinned here.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from cowork_local.application.conversations.cowork_turn_request import (
|
||||
build_cowork_turn_request,
|
||||
)
|
||||
|
||||
|
||||
def _build(**overrides):
|
||||
base = {
|
||||
"turn_id": "t3",
|
||||
"session_id": "s1",
|
||||
"messages": [{"role": "user", "content": "make me a report"}],
|
||||
}
|
||||
base.update(overrides)
|
||||
return build_cowork_turn_request(**base)
|
||||
|
||||
|
||||
def test_the_last_message_becomes_the_prompt_and_the_rest_the_history() -> None:
|
||||
request = _build(messages=[
|
||||
{"role": "user", "content": "earlier"},
|
||||
{"role": "assistant", "content": "sure"},
|
||||
{"role": "user", "content": "now this"},
|
||||
])
|
||||
|
||||
assert request.prompt == "now this"
|
||||
assert request.messages == ({"role": "user", "content": "earlier"},
|
||||
{"role": "assistant", "content": "sure"})
|
||||
|
||||
|
||||
def test_an_empty_working_list_yields_an_empty_prompt() -> None:
|
||||
# Defensive: a turn with no message at all must not raise on messages[-1].
|
||||
request = _build(messages=[])
|
||||
|
||||
assert request.prompt == ""
|
||||
assert request.messages == ()
|
||||
|
||||
|
||||
def test_confirming_commands_puts_the_turn_in_confirm_gate_mode() -> None:
|
||||
assert _build(confirm_commands=True).gate_mode == "confirm"
|
||||
assert _build(confirm_commands=False).gate_mode == "auto"
|
||||
assert _build().gate_mode == "auto" # auto-run is the default
|
||||
|
||||
|
||||
def test_the_captured_widget_state_is_carried_into_the_request() -> None:
|
||||
request = _build(
|
||||
surface="cowork", project_id="p7", title="Weekly report",
|
||||
provider_id="anthropic", model="claude-sonnet-4-6",
|
||||
instructions="PROJECT RULES", output_dir="out/.turns/t3",
|
||||
home_output_root="out", agent_role="cowork",
|
||||
)
|
||||
|
||||
assert (request.turn_id, request.session_id) == ("t3", "s1")
|
||||
assert (request.surface, request.project_id, request.title) == \
|
||||
("cowork", "p7", "Weekly report")
|
||||
assert (request.provider_id, request.model) == ("anthropic", "claude-sonnet-4-6")
|
||||
assert request.project_context == "PROJECT RULES"
|
||||
assert request.output_dir == Path("out/.turns/t3")
|
||||
assert request.home_output_root == Path("out")
|
||||
assert request.agent_role == "cowork"
|
||||
|
||||
|
||||
def test_the_prompt_survives_a_message_whose_content_is_missing() -> None:
|
||||
request = _build(messages=[{"role": "user"}])
|
||||
|
||||
assert request.prompt == ""
|
||||
@@ -0,0 +1,34 @@
|
||||
"""R04-T05 — unit tests for the unattended-run prompt assembly.
|
||||
|
||||
``_run_agent`` used to build this by rebinding ``prompt`` three times, each with
|
||||
its own ``f"{block}\n\n{prompt}"``. The ORDER that produced is load-bearing (the
|
||||
plan reminder has to lead, the task's own words have to trail) and it was
|
||||
readable only by replaying the rebindings in your head.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from cowork_local.core.task_executors import _unattended_prompt
|
||||
|
||||
|
||||
def test_the_plan_reminder_leads_and_the_task_prompt_trails() -> None:
|
||||
built = _unattended_prompt("write the report")
|
||||
|
||||
assert built.startswith("This runs unattended (Schedule Task)")
|
||||
assert built.endswith("write the report")
|
||||
|
||||
|
||||
def test_a_skill_block_sits_between_the_reminder_and_the_agent_persona() -> None:
|
||||
built = _unattended_prompt("write the report", skill_text="SKILL",
|
||||
agent_instructions="PERSONA")
|
||||
|
||||
assert built.index("This runs unattended") < built.index("SKILL")
|
||||
assert built.index("SKILL") < built.index("PERSONA")
|
||||
assert built.index("PERSONA") < built.index("write the report")
|
||||
|
||||
|
||||
def test_absent_blocks_leave_no_extra_blank_lines() -> None:
|
||||
built = _unattended_prompt("do it", skill_text="", agent_instructions=None)
|
||||
|
||||
assert "\n\n\n" not in built
|
||||
assert built.count("do it") == 1
|
||||
@@ -0,0 +1,36 @@
|
||||
"""R04-T04 — unit tests for the shared turn-runtime helpers.
|
||||
|
||||
``combine_instructions`` is the small rule the UI applied inline: a turn's
|
||||
standing instructions are several independent blocks (project context, an Admin
|
||||
agent's persona, a skill's rules, an unattended-run reminder) that must be joined
|
||||
with one blank line, skipping whatever is absent. Two call sites need it (T04's
|
||||
widget and T05's task runner), which is exactly when a rule stops being an inline
|
||||
expression.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from cowork_local.application.conversations.turn_runtime import combine_instructions
|
||||
|
||||
|
||||
def test_two_blocks_are_joined_by_a_blank_line() -> None:
|
||||
assert combine_instructions("PROJECT", "AGENT") == "PROJECT\n\nAGENT"
|
||||
|
||||
|
||||
def test_an_absent_block_leaves_no_blank_line_behind() -> None:
|
||||
assert combine_instructions("", "AGENT") == "AGENT"
|
||||
assert combine_instructions("PROJECT", "") == "PROJECT"
|
||||
assert combine_instructions("PROJECT", None) == "PROJECT"
|
||||
|
||||
|
||||
def test_whitespace_only_blocks_do_not_count_as_instructions() -> None:
|
||||
assert combine_instructions(" \n ", "AGENT") == "AGENT"
|
||||
|
||||
|
||||
def test_nothing_to_say_produces_an_empty_string() -> None:
|
||||
assert combine_instructions() == ""
|
||||
assert combine_instructions("", None, " ") == ""
|
||||
|
||||
|
||||
def test_more_than_two_blocks_keep_their_order() -> None:
|
||||
assert combine_instructions("A", "B", "C") == "A\n\nB\n\nC"
|
||||
Reference in New Issue
Block a user