feat(R04): run every Cowork turn through ConversationApplicationService
R04-T03 — the turn lifecycle, extracted from `core/chat_agent.py::run_cowork` into `application/conversations/`. The 260-line body mixed the lifecycle (step budget, cancel checks, guard -> preview -> gate -> execute ordering, sandbox tidy-up) with the machinery doing each step, and reaching any of it meant standing up a Qt widget and a worker thread. It is now a plain object driven through two Protocols and six callables (`turn_runtime.py`), with the concrete `core/*` wiring confined to `core_runtime_adapter.py` — the same shape R03 used for routing. Faithful port, not an improvement pass: where the original had a quirk (the step-ceiling note only merges into the answer when the last message is the assistant's) the quirk is preserved and commented. R04-T04 — `ui/cowork_tab.py::build_job` no longer calls run_cowork. It captures the widget's state at submit time, builds the request via the new `cowork_turn_request.py` and executes it. `execute(..., messages=...)` hands the widget's own list over because `_reattach_running_turn` replays from it WHILE the worker appends and `_finalize_turn` slices it afterwards — a private list would break both silently. R04-T05 — `core/task_executors.py`'s cowork branch shares the same engine. All five unattended-run behaviours stay put (plan reminder, history_ready, History autosave per assistant message, timeout notice, plan_incomplete_reason), and `_unattended_prompt` now expresses the load-bearing prefix order in one readable call instead of three successive rebindings. Verification: 74 new tests (364 passed, 1 skipped overall; check_imports PASS). The two that matter most: - `test_conversation_service_parity.py` runs the same scripted turn through run_cowork AND the service and compares the event stream, the resulting conversation and the advertised tool list across 7 scenarios; - `test_task_executor_turn.py` was written BEFORE the migration and passed 8/8 against the old code, then unchanged against the new. Known: `ui/cowork_tab.py` (416 -> 455) and `core/task_executors.py` (476 -> 524) stay above the 400-LOC limit. Both were already over it before this change; bringing them under needs the R08 / R07 decompositions. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,322 @@
|
||||
"""The turn lifecycle, once, in pure Python (R04-T03).
|
||||
|
||||
Extracted from ``core/chat_agent.py::run_cowork``, whose 260-line body mixed the
|
||||
lifecycle (compose the prompt, call the model, dispatch tools, respect the step
|
||||
ceiling, tidy the sandbox) with the concrete machinery that does each of those
|
||||
things. The lifecycle is the part with rules worth testing — and the part that
|
||||
was untestable, because reaching it meant standing up a Qt widget and a worker
|
||||
thread.
|
||||
|
||||
Here it is a plain object driven through the seams in :mod:`turn_runtime`, so a
|
||||
test states a rule ("the guard runs before the model", "a rejected command never
|
||||
executes") in three lines. ``core/chat_agent.py`` keeps its signature and
|
||||
delegates, and the presentation layer keeps receiving the same events via the
|
||||
legacy codec, so nothing downstream had to change with it.
|
||||
|
||||
Behavioural contract: this is a faithful port, not an improvement pass. Where
|
||||
the original had a quirk (the step-ceiling note only merges into the answer when
|
||||
the last message is the assistant's), the quirk is preserved and commented —
|
||||
changing what a user sees belongs in its own change, not smuggled into a move.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
from ...domain.agents.agent_event import (
|
||||
AssistantMessageCompletedEvent,
|
||||
ErrorEvent,
|
||||
OutputsAddedEvent,
|
||||
OutputsRemovedEvent,
|
||||
PlanStep,
|
||||
PlanUpdatedEvent,
|
||||
ReasoningChunkEvent,
|
||||
TextChunkEvent,
|
||||
ToolCallFinishedEvent,
|
||||
ToolCallStartedEvent,
|
||||
ToolOutputChunkEvent,
|
||||
)
|
||||
from ...domain.agents.agent_result import AgentResult
|
||||
from ...domain.agents.conversation_execution_request import ConversationExecutionRequest
|
||||
from .turn_runtime import (
|
||||
BUDGET_NOTE_TEMPLATE,
|
||||
GATED_TOOLS,
|
||||
PLAN_TOOL,
|
||||
REASONING_ONLY_NOTE,
|
||||
REJECTED_OUTPUT,
|
||||
AttachmentReader,
|
||||
CancelFn,
|
||||
CommandGuard,
|
||||
ContextCompactor,
|
||||
EventSink,
|
||||
ModelCallPort,
|
||||
PermissionRequest,
|
||||
PromptGuard,
|
||||
PromptPreparer,
|
||||
ToolRuntimePort,
|
||||
)
|
||||
|
||||
logger = logging.getLogger("cowork_local.application.conversations")
|
||||
|
||||
|
||||
class ConversationApplicationService:
|
||||
"""Runs one :class:`ConversationExecutionRequest` to completion."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
model: ModelCallPort,
|
||||
tools: ToolRuntimePort,
|
||||
*,
|
||||
prepare_prompt: Optional[PromptPreparer] = None,
|
||||
prompt_guard: Optional[PromptGuard] = None,
|
||||
command_guard: Optional[CommandGuard] = None,
|
||||
compact: Optional[ContextCompactor] = None,
|
||||
permission_request: Optional[PermissionRequest] = None,
|
||||
attachment_reader: Optional[AttachmentReader] = None,
|
||||
) -> None:
|
||||
self._model = model
|
||||
self._tools = tools
|
||||
# Every hook is optional so the service degrades to a plain chat turn.
|
||||
# That is not only a test convenience: a headless caller legitimately has
|
||||
# no guards (``security_config=None`` today) and no permission dialog.
|
||||
self._prepare_prompt = prepare_prompt
|
||||
self._prompt_guard = prompt_guard
|
||||
self._command_guard = command_guard
|
||||
self._compact = compact
|
||||
self._permission_request = permission_request
|
||||
self._attachment_reader = attachment_reader
|
||||
|
||||
# -- public API ------------------------------------------------------ #
|
||||
def execute(self, request: ConversationExecutionRequest, sink: EventSink,
|
||||
cancel: Optional[CancelFn] = None,
|
||||
messages: Optional[List[Dict[str, Any]]] = None) -> AgentResult:
|
||||
"""Run the turn, streaming events to ``sink``, and report the outcome.
|
||||
|
||||
``messages``, when given, is a working list the caller already built —
|
||||
it MUST already end with this turn's user message, and the service
|
||||
appends into that very object instead of composing its own. The Cowork
|
||||
widget needs this: it hands out the same list to
|
||||
``_reattach_running_turn``, which replays the steps done so far while the
|
||||
worker is still appending, and to ``_finalize_turn``, which slices it by
|
||||
the pre-turn snapshot length. A private list would break both silently.
|
||||
Passing ``None`` (every headless caller) lets the service compose the
|
||||
list from the request, which is the mode the rest of this class assumes.
|
||||
|
||||
Raises whatever the runtime raises (a blocked prompt, a dead gateway):
|
||||
the caller already has a failure path for that — ``AgentWorker.failed``
|
||||
in the UI, the artifact writer in Schedule Task — and swallowing the
|
||||
exception here would silently turn a failed turn into an empty answer.
|
||||
An :class:`ErrorEvent` is emitted first so subscribers see the failure
|
||||
on the same stream as everything else.
|
||||
"""
|
||||
cancel = cancel or (lambda: False)
|
||||
|
||||
# -- pre-flight. Runs BEFORE the output snapshot, so a turn refused here
|
||||
# leaves the output folder completely untouched (tidying is not a
|
||||
# read-only operation — see ToolRuntimePort.finalize).
|
||||
try:
|
||||
# The caller's list is used by reference on purpose (see above); only
|
||||
# the self-composed path may build a fresh one.
|
||||
working = messages if messages is not None else self._compose_messages(request)
|
||||
tools = list(self._tools.specs(request.allowed_tools))
|
||||
if self._prepare_prompt is not None:
|
||||
self._prepare_prompt(working, tuple(getattr(t, "name", "") for t in tools))
|
||||
if request.enforce_rules and self._prompt_guard is not None:
|
||||
self._prompt_guard(working)
|
||||
except Exception as exc: # noqa: BLE001 — reported, then re-raised as-is
|
||||
sink(ErrorEvent(message=str(exc)))
|
||||
raise
|
||||
|
||||
before = self._tools.snapshot()
|
||||
steps_used = 0
|
||||
plan_steps: Tuple[PlanStep, ...] = ()
|
||||
completed_naturally = False
|
||||
try:
|
||||
for _ in range(request.effective_max_steps):
|
||||
if cancel():
|
||||
break
|
||||
# Auto-compress when nearing the model's context budget; a no-op
|
||||
# when off or when the conversation is still short.
|
||||
if self._compact is not None:
|
||||
self._compact(working, cancel)
|
||||
|
||||
assistant = self._model.call(
|
||||
working, tools,
|
||||
on_text=lambda piece: sink(TextChunkEvent(delta=piece)),
|
||||
on_reasoning=lambda piece: sink(ReasoningChunkEvent(delta=piece)),
|
||||
cancel=cancel,
|
||||
)
|
||||
working.append(assistant)
|
||||
steps_used += 1
|
||||
tool_calls = assistant.get("tool_calls") or []
|
||||
|
||||
if not tool_calls and not (assistant.get("content") or "").strip():
|
||||
# Written into the message, not just emitted, so the stored
|
||||
# conversation never ends on a blank assistant turn.
|
||||
assistant["content"] = REASONING_ONLY_NOTE
|
||||
sink(TextChunkEvent(delta=REASONING_ONLY_NOTE))
|
||||
sink(AssistantMessageCompletedEvent(content=assistant.get("content", "")))
|
||||
|
||||
if not tool_calls:
|
||||
completed_naturally = True
|
||||
break
|
||||
|
||||
for call in tool_calls:
|
||||
if cancel():
|
||||
break
|
||||
tool_message, steps = self._dispatch(request, call, sink, cancel)
|
||||
working.append(tool_message)
|
||||
if steps is not None:
|
||||
plan_steps = steps
|
||||
|
||||
if not completed_naturally and not cancel():
|
||||
self._announce_budget_exhausted(request, working, sink)
|
||||
except Exception as exc: # noqa: BLE001 — reported, then re-raised as-is
|
||||
sink(ErrorEvent(message=str(exc)))
|
||||
raise
|
||||
finally:
|
||||
# Always tidy: the sandbox and generator scripts must not survive a
|
||||
# turn that stopped abruptly. Runs on success, cancel and failure.
|
||||
self._finalize_outputs(before, sink, cancelled=cancel())
|
||||
|
||||
result = AgentResult(
|
||||
messages=working, steps_used=steps_used, cancelled=cancel(),
|
||||
budget_exhausted=not completed_naturally and not cancel(),
|
||||
plan_steps=plan_steps,
|
||||
)
|
||||
sink(result.to_turn_completed_event())
|
||||
return result
|
||||
|
||||
# -- internals ------------------------------------------------------- #
|
||||
def _compose_messages(self, request: ConversationExecutionRequest) -> List[Dict[str, Any]]:
|
||||
"""History snapshot plus this turn's user message.
|
||||
|
||||
The attachment text is read HERE rather than when the request was built,
|
||||
because extraction is slow enough to freeze the UI thread; the request
|
||||
deliberately carries paths only.
|
||||
"""
|
||||
body = request.prompt
|
||||
if self._attachment_reader is not None:
|
||||
body = self._attachment_reader(request.prompt, request.attachments)
|
||||
messages = [dict(m) for m in request.messages]
|
||||
messages.append({"role": "user", "content": request.user_content(body)})
|
||||
return messages
|
||||
|
||||
def _dispatch(self, request: ConversationExecutionRequest, call: Dict[str, Any],
|
||||
sink: EventSink, cancel: CancelFn
|
||||
) -> Tuple[Dict[str, Any], Optional[Tuple[PlanStep, ...]]]:
|
||||
"""Run one tool call.
|
||||
|
||||
Returns ``(tool_message, plan_steps)`` — the message to append to the
|
||||
conversation, and the new checklist when this call was the plan tool
|
||||
(``None`` otherwise, so the caller can tell "no change" from "empty
|
||||
plan").
|
||||
"""
|
||||
call_id = str(call.get("id", ""))
|
||||
name = str(call.get("name", ""))
|
||||
args = call.get("arguments") or {}
|
||||
|
||||
# The plan tool is invisible in the transcript: it updates the Plan panel
|
||||
# and nothing else, so it skips preview, guard and gate entirely.
|
||||
if name == PLAN_TOOL:
|
||||
outcome = self._tools.execute(name, args, on_output=None, cancel=cancel)
|
||||
steps = tuple(outcome.get("plan_steps") or ())
|
||||
sink(PlanUpdatedEvent(steps=steps))
|
||||
return self._tool_message(call_id, name, outcome.get("output", "")), steps
|
||||
|
||||
# Announce first: the user sees the code/command about to run before the
|
||||
# guard or the approval dialog interrupts them, which is the whole point
|
||||
# of showing the step CLI-style.
|
||||
preview = self._tools.preview(name, args)
|
||||
sink(ToolCallStartedEvent(call_id=call_id, name=name, arguments=dict(args),
|
||||
preview=preview))
|
||||
|
||||
if request.enforce_rules and self._command_guard is not None:
|
||||
self._command_guard(name, args)
|
||||
|
||||
if not self._approved(request, name, args, preview, sink, call_id):
|
||||
return self._tool_message(call_id, name, REJECTED_OUTPUT), None
|
||||
|
||||
outcome = self._tools.execute(
|
||||
name, args,
|
||||
on_output=lambda piece: sink(ToolOutputChunkEvent(
|
||||
call_id=call_id, name=name, delta=piece)),
|
||||
cancel=cancel,
|
||||
)
|
||||
sink(ToolCallFinishedEvent(
|
||||
call_id=call_id, name=name, ok=bool(outcome.get("ok", False)),
|
||||
output=str(outcome.get("output", "")), path=str(outcome.get("path", "") or ""),
|
||||
produced=outcome.get("produced") or (),
|
||||
))
|
||||
return self._tool_message(call_id, name, outcome.get("output", "")), None
|
||||
|
||||
def _approved(self, request: ConversationExecutionRequest, name: str,
|
||||
args: Dict[str, Any], preview: Any, sink: EventSink,
|
||||
call_id: str) -> bool:
|
||||
"""Whether this call may run.
|
||||
|
||||
Only command-shaped tools are gated, and only when the workspace asked
|
||||
to confirm them: file writes stay inside the turn's own sandbox, so
|
||||
prompting for those would be noise. A rejection is reported as a failed
|
||||
tool result — the model needs to read back that it was refused, or it
|
||||
will simply try the same call again.
|
||||
"""
|
||||
if not request.requires_permission_gate or name not in GATED_TOOLS:
|
||||
return True
|
||||
if self._permission_request is None:
|
||||
# Confirm mode with nobody to ask: refusing is the safe direction,
|
||||
# since auto-running is exactly what confirm mode exists to prevent.
|
||||
logger.warning("turn: confirm mode without a permission callback — refusing %r", name)
|
||||
approved = False
|
||||
else:
|
||||
approved = bool(self._permission_request({
|
||||
"name": name, "args": args,
|
||||
"preview": preview.to_dict() if preview is not None else {},
|
||||
}))
|
||||
if not approved:
|
||||
sink(ToolCallFinishedEvent(call_id=call_id, name=name, ok=False,
|
||||
output=REJECTED_OUTPUT))
|
||||
return approved
|
||||
|
||||
@staticmethod
|
||||
def _tool_message(call_id: str, name: str, output: Any) -> Dict[str, Any]:
|
||||
"""The canonical ``role: tool`` message the model reads back."""
|
||||
return {"role": "tool", "tool_call_id": call_id, "name": name,
|
||||
"content": str(output or "")}
|
||||
|
||||
@staticmethod
|
||||
def _announce_budget_exhausted(request: ConversationExecutionRequest,
|
||||
messages: List[Dict[str, Any]], sink: EventSink) -> None:
|
||||
"""Report being cut off by the step ceiling.
|
||||
|
||||
The note always reaches the transcript. It is merged into the stored
|
||||
answer only when the last message is the assistant's — which, when the
|
||||
ceiling is hit, it never is (the turn ends on a tool result). The branch
|
||||
is kept because it is what the current runtime does, and because it is
|
||||
the correct behaviour the day a caller ends the loop differently.
|
||||
"""
|
||||
note = BUDGET_NOTE_TEMPLATE.format(steps=request.effective_max_steps)
|
||||
sink(TextChunkEvent(delta=note))
|
||||
if messages and messages[-1].get("role") == "assistant":
|
||||
messages[-1]["content"] = (messages[-1].get("content") or "") + note
|
||||
|
||||
def _finalize_outputs(self, before: Any, sink: EventSink, cancelled: bool) -> None:
|
||||
"""Tidy the output folder and report what moved.
|
||||
|
||||
Failures are logged, never raised: this runs in a ``finally``, so an
|
||||
exception here would replace the turn's real error (or its success) with
|
||||
a housekeeping one.
|
||||
"""
|
||||
try:
|
||||
removed, added = self._tools.finalize(before, cancelled=cancelled)
|
||||
except Exception: # noqa: BLE001
|
||||
logger.exception("turn: tidying the output folder failed")
|
||||
return
|
||||
if removed:
|
||||
sink(OutputsRemovedEvent(paths=tuple(removed)))
|
||||
if added:
|
||||
sink(OutputsAddedEvent(paths=tuple(added)))
|
||||
|
||||
|
||||
__all__ = ["ConversationApplicationService"]
|
||||
@@ -0,0 +1,325 @@
|
||||
"""Wires :class:`ConversationApplicationService` to the existing runtime (R04-T03).
|
||||
|
||||
The service is written against the narrow seams in :mod:`turn_runtime` so it can
|
||||
be tested with plain fakes. This module supplies the real implementations — the
|
||||
provider call with its recovery pass, the tool/sandbox runtime, the security
|
||||
guards, context compaction — and is therefore the ONLY file in
|
||||
``application/conversations/`` that knows ``core/*`` exists. Same shape (and
|
||||
same reason) as ``application/model_routing/core_routing_adapter.py`` in R03.
|
||||
|
||||
Every ``core`` import is deferred into a method body: importing the tool runtime
|
||||
pulls in ``requests``, ``psutil`` and the sandbox stack, and code that merely
|
||||
*builds* a service must not pay for that.
|
||||
|
||||
Faithfulness notes — two places where this reproduces a quirk of the current
|
||||
runtime rather than the behaviour one would design fresh. Both are marked
|
||||
inline: the MS365 system-prompt paragraph keys off the CONFIGURED extra tools
|
||||
(not the advertised subset), and the ``tool_result`` path falls back to the
|
||||
call's own ``path`` argument resolved against the workdir.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable, Dict, List, Optional, Sequence, Tuple
|
||||
|
||||
from ...domain.agents.agent_event import PlanStep, ToolPreview
|
||||
from .conversation_application_service import ConversationApplicationService
|
||||
from .turn_runtime import PLAN_TOOL, EventSink
|
||||
|
||||
# Legacy emit: the dict-based callback every current caller already owns.
|
||||
LegacyEmit = Callable[[Dict[str, Any]], None]
|
||||
|
||||
|
||||
def legacy_event_sink(emit: LegacyEmit) -> EventSink:
|
||||
"""Adapt a typed :class:`EventSink` onto the legacy dict ``emit``.
|
||||
|
||||
This is what lets R04 land without touching the presentation layer: the
|
||||
service thinks in typed events, ``ui/chat_panel.py::_on_event`` keeps
|
||||
receiving exactly the dicts it already dispatches on. Deleted in R08 once
|
||||
the widget consumes events directly.
|
||||
"""
|
||||
return lambda event: emit(event.to_legacy_dict())
|
||||
|
||||
|
||||
class CoreModelCall:
|
||||
""":class:`ModelCallPort` over ``code_agent._call_provider_with_recovery``.
|
||||
|
||||
Not ``provider.chat`` directly: the recovery wrapper adds the one bounded
|
||||
retry that hides a dropped connection or a momentarily unreachable gateway,
|
||||
and losing it would be a visible regression on flaky corporate networks.
|
||||
"""
|
||||
|
||||
def __init__(self, provider: Any) -> None:
|
||||
self._provider = provider
|
||||
|
||||
def call(self, messages, tools, on_text=None, on_reasoning=None, cancel=None):
|
||||
from ...core.code_agent import _call_provider_with_recovery
|
||||
|
||||
return _call_provider_with_recovery(self._provider, messages, tools, on_text,
|
||||
cancel, on_reasoning)
|
||||
|
||||
|
||||
class CoreToolRuntime:
|
||||
""":class:`ToolRuntimePort` over ``core/tools.py`` + Cowork's file tools."""
|
||||
|
||||
def __init__(self, output_dir: Path, *, title: str = "",
|
||||
extra_tools: Optional[Sequence[Any]] = None, extra_executor=None,
|
||||
security_config: Any = None, agent_role: str = "") -> None:
|
||||
self._output_dir = Path(output_dir)
|
||||
self._title = title
|
||||
self._extra_tools = list(extra_tools or ())
|
||||
self._extra_names = {getattr(t, "name", "") for t in self._extra_tools}
|
||||
# The connector executor MCP/REST tools are routed to; None when the
|
||||
# turn has no connectors enabled.
|
||||
self._extra_executor = extra_executor
|
||||
self._security_config = security_config
|
||||
self._agent_role = agent_role
|
||||
self._ctx: Any = None # built on first use (see _tool_context)
|
||||
|
||||
# -- the configured extra tools, for the system-prompt hints ---------- #
|
||||
@property
|
||||
def extra_names(self) -> frozenset:
|
||||
return frozenset(self._extra_names)
|
||||
|
||||
def _tool_context(self):
|
||||
"""The sandboxed ``ToolContext`` every built-in tool call runs inside.
|
||||
|
||||
Built once per turn and cached: it carries the resource limits and the
|
||||
network policy, so re-deriving it mid-turn could let a Settings change
|
||||
take effect halfway through work already in flight.
|
||||
"""
|
||||
if self._ctx is None:
|
||||
from ...core import agent_security
|
||||
from ...core.tools import ToolContext
|
||||
|
||||
limits, block_network = agent_security.sandbox_settings(self._security_config)
|
||||
self._ctx = ToolContext(
|
||||
self._output_dir, flatten_writes=True, # keep every file in the Output root
|
||||
resource_limits=limits, block_network=block_network,
|
||||
allow_url_fetch=agent_security.url_fetch_allowed(self._security_config),
|
||||
jira=(self._security_config.data.get("jira") if self._security_config else None),
|
||||
)
|
||||
return self._ctx
|
||||
|
||||
# -- ToolRuntimePort -------------------------------------------------- #
|
||||
def specs(self, allowed_tools: Optional[Sequence[str]] = None) -> List[Any]:
|
||||
"""Advertised tools: Cowork's own two, the enabled built-ins, then MCP.
|
||||
|
||||
``allowed_tools`` restricts the list so a read-only step literally cannot
|
||||
write. ``update_plan`` and the connector tools always survive the filter:
|
||||
the plan tool has no side effects, and connectors are opted into
|
||||
explicitly rather than governed by the built-in capability scope.
|
||||
"""
|
||||
from ...core.chat_agent import SAVE_FILE_SPEC
|
||||
from ...core.plan import UPDATE_PLAN_SPEC
|
||||
from ...core.tools import enabled_tool_specs
|
||||
|
||||
specs = ([SAVE_FILE_SPEC, UPDATE_PLAN_SPEC]
|
||||
+ list(enabled_tool_specs(self._security_config))
|
||||
+ self._extra_tools)
|
||||
if allowed_tools is None:
|
||||
return specs
|
||||
allow = set(allowed_tools) | {PLAN_TOOL} | self._extra_names
|
||||
return [t for t in specs if getattr(t, "name", "") in allow]
|
||||
|
||||
def preview(self, name: str, args: Dict[str, Any]) -> Optional[ToolPreview]:
|
||||
"""What the user sees before the call runs."""
|
||||
# A connector call has no local diff to show, so it renders as the plain
|
||||
# argument dump the runtime already used.
|
||||
if name in self._extra_names:
|
||||
return ToolPreview(kind="info", title=name, text=str(args))
|
||||
if name == "save_file":
|
||||
return self._save_file_preview(args)
|
||||
from ...core.tools import describe_action
|
||||
|
||||
raw = describe_action(self._tool_context(), name, args)
|
||||
return ToolPreview.from_dict(raw)
|
||||
|
||||
def _save_file_preview(self, args: Dict[str, Any]) -> ToolPreview:
|
||||
"""A before/after diff for the file the agent is about to write.
|
||||
|
||||
A brand-new file renders all-green (before is empty); an overwrite shows
|
||||
the real change, so saving a file reads like editing one.
|
||||
"""
|
||||
import difflib
|
||||
|
||||
from ...core.chat_agent import _structure_summary, _titled_filename
|
||||
|
||||
fname = _titled_filename(self._title, args.get("filename", "output.txt"))
|
||||
content = str(args.get("content", ""))
|
||||
summary = _structure_summary(fname, content)
|
||||
old = ""
|
||||
existing = self._output_dir / fname
|
||||
if existing.exists():
|
||||
try:
|
||||
old = existing.read_text(encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
pass # unreadable existing file: show it as a fresh write
|
||||
diff = "".join(difflib.unified_diff(
|
||||
old.splitlines(keepends=True), content.splitlines(keepends=True),
|
||||
fromfile=f"a/{fname}", tofile=f"b/{fname}",
|
||||
)) or content[:4000]
|
||||
return ToolPreview(kind="diff", title=f"Save {fname}",
|
||||
text=f"{summary}\n\n{diff[:4000]}")
|
||||
|
||||
def execute(self, name: str, args: Dict[str, Any], on_output=None,
|
||||
cancel=None) -> Dict[str, Any]:
|
||||
"""Run one tool call and return the runtime's result mapping."""
|
||||
if name == PLAN_TOOL:
|
||||
return self._execute_plan(args)
|
||||
if name in self._extra_names and self._extra_executor is not None:
|
||||
# Connector results carry no local file, so no path/produced keys —
|
||||
# matching what the runtime reports for an MCP call today.
|
||||
result = self._extra_executor(name, args) or {}
|
||||
return {"ok": bool(result.get("ok", False)), "output": result.get("output", "")}
|
||||
if name == "save_file":
|
||||
from ...core.chat_agent import _do_save_file
|
||||
|
||||
return dict(_do_save_file(self._output_dir, self._title, args))
|
||||
|
||||
from ...core.tools import execute_tool
|
||||
|
||||
ctx = self._tool_context()
|
||||
result = dict(execute_tool(ctx, name, args, cancel=cancel, on_output=on_output,
|
||||
agent_role=self._agent_role))
|
||||
# Quirk preserved: a tool that wrote the file named in its OWN arguments
|
||||
# (write_file/edit_file) does not report a path, so the runtime derives
|
||||
# one from the argument. Dropping this would empty the Output list.
|
||||
if not result.get("path") and isinstance(args, dict) and args.get("path"):
|
||||
result["path"] = str(ctx.workdir / str(args["path"]))
|
||||
return result
|
||||
|
||||
def _execute_plan(self, args: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Apply an ``update_plan`` call: validate the steps and audit them.
|
||||
|
||||
Produces no file and no chat bubble; the service turns the returned
|
||||
steps into a single plan event.
|
||||
"""
|
||||
from ...core import agent_roles, audit_log
|
||||
from ...core.plan import normalize_plan_steps
|
||||
|
||||
steps = normalize_plan_steps(args.get("steps"))
|
||||
audit_log.record("tool_call", PLAN_TOOL, True, f"{len(steps)} step(s)",
|
||||
agent_role=agent_roles.PLANNER)
|
||||
return {"ok": True, "output": "Plan updated.",
|
||||
"plan_steps": [PlanStep(title=s["title"], status=s["status"]) for s in steps]}
|
||||
|
||||
def snapshot(self) -> Any:
|
||||
from ...core.tools import _snapshot
|
||||
|
||||
return _snapshot(self._output_dir)
|
||||
|
||||
def finalize(self, before: Any, cancelled: bool = False
|
||||
) -> Tuple[List[str], List[str]]:
|
||||
"""Drop the scratch sandbox and flatten deliverables into the root.
|
||||
|
||||
Returns ``(gone, arrived)``: a file that MOVED counts as both, because
|
||||
the Output list keys entries by path and must drop the old one.
|
||||
"""
|
||||
from ...core.chat_agent import _cleanup_cowork_intermediates
|
||||
|
||||
removed, moved = _cleanup_cowork_intermediates(self._output_dir, before,
|
||||
cancelled=cancelled)
|
||||
gone = list(removed) + [old for old, _new in moved]
|
||||
arrived = [new for _old, new in moved]
|
||||
return gone, arrived
|
||||
|
||||
|
||||
def build_cowork_conversation_service(
|
||||
provider: Any,
|
||||
output_dir: Path,
|
||||
emit: LegacyEmit,
|
||||
*,
|
||||
title: str = "",
|
||||
project_context: str = "",
|
||||
extra_tools: Optional[Sequence[Any]] = None,
|
||||
extra_executor=None,
|
||||
security_config: Any = None,
|
||||
gate: Any = None,
|
||||
agent_role: str = "",
|
||||
) -> ConversationApplicationService:
|
||||
"""A service wired to the real runtime, ready to execute a Cowork turn.
|
||||
|
||||
``emit`` is the legacy dict callback: the guards and the compactor publish
|
||||
their own notices through it directly (exactly as they do now), while the
|
||||
service's typed events reach it via :func:`legacy_event_sink`.
|
||||
|
||||
``gate`` present means the workspace asked to confirm commands; pass the
|
||||
request with ``gate_mode="confirm"`` so the two agree. A gate of ``None``
|
||||
keeps the pre-existing auto-run behaviour.
|
||||
"""
|
||||
from ...core import agent_roles
|
||||
|
||||
tools = CoreToolRuntime(
|
||||
output_dir, title=title, extra_tools=extra_tools, extra_executor=extra_executor,
|
||||
security_config=security_config, agent_role=agent_role or agent_roles.COWORK,
|
||||
)
|
||||
|
||||
def prepare_prompt(messages: List[Dict[str, Any]], advertised: Tuple[str, ...]) -> None:
|
||||
"""Insert the system prompt, then fold in skills, rules and project text.
|
||||
|
||||
``advertised`` is unused on purpose: the runtime decides the MS365
|
||||
paragraph from the CONFIGURED connector tools, not from the subset a
|
||||
capability scope left advertised. Changing that changes the prompt the
|
||||
model sees, so it stays as-is here and belongs to R05's tool-policy work.
|
||||
"""
|
||||
from ...core.chat_agent import (
|
||||
COWORK_TOOL_PROMPT,
|
||||
OPENDATALOADER_PDF_PROMPT,
|
||||
_apply_project_context,
|
||||
_apply_security_rules,
|
||||
_apply_skills,
|
||||
)
|
||||
from ...core.deps import _can_pip
|
||||
from ...core.java_runtime import find_java
|
||||
from ...core.security_rules import load_rules
|
||||
from ...core.skills import active_skills_text
|
||||
|
||||
if not messages or messages[0].get("role") != "system":
|
||||
system = COWORK_TOOL_PROMPT
|
||||
if any(n.startswith("ms365_") for n in tools.extra_names):
|
||||
system += ("\nThe user has signed in to Microsoft 365 and enabled some ms365__* "
|
||||
"tools (Outlook / Teams / OneDrive / SharePoint / meeting transcripts, "
|
||||
"via the built-in MS365 MCP server). Use them whenever the request "
|
||||
"involves that data — don't say you can't access it.")
|
||||
if find_java() is not None and _can_pip():
|
||||
# Only advertise the Java-backed PDF extractor when BOTH the JVM
|
||||
# and pip are available, so the agent is never steered into a
|
||||
# command that cannot work on this machine.
|
||||
system += "\n\n" + OPENDATALOADER_PDF_PROMPT
|
||||
messages.insert(0, {"role": "system", "content": system})
|
||||
_apply_skills(messages, active_skills_text())
|
||||
_apply_security_rules(messages, load_rules())
|
||||
_apply_project_context(messages, project_context)
|
||||
|
||||
def prompt_guard(messages: List[Dict[str, Any]]) -> None:
|
||||
from ...core import agent_security
|
||||
|
||||
agent_security.enforce_prompt(provider, messages, security_config, emit)
|
||||
|
||||
def command_guard(name: str, args: Dict[str, Any]) -> None:
|
||||
from ...core import agent_security
|
||||
|
||||
agent_security.enforce_command(provider, name, args, security_config, emit)
|
||||
|
||||
def compact(messages: List[Dict[str, Any]], cancel) -> None:
|
||||
from ...core import context_budget
|
||||
|
||||
context_budget.maybe_compact(provider, messages, security_config,
|
||||
emit=emit, cancel=cancel)
|
||||
|
||||
return ConversationApplicationService(
|
||||
CoreModelCall(provider), tools,
|
||||
prepare_prompt=prepare_prompt,
|
||||
prompt_guard=prompt_guard,
|
||||
command_guard=command_guard,
|
||||
compact=compact,
|
||||
permission_request=(gate.request if gate is not None else None),
|
||||
)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"LegacyEmit", "legacy_event_sink", "CoreModelCall", "CoreToolRuntime",
|
||||
"build_cowork_conversation_service",
|
||||
]
|
||||
@@ -0,0 +1,77 @@
|
||||
"""Turn the Cowork widget's captured state into a request (R04-T04).
|
||||
|
||||
``ui/cowork_tab.py::build_job`` reads a dozen values off the widget on the UI
|
||||
thread and has to translate three of them before a turn can run: which message
|
||||
is this turn's prompt, which messages are its history, and whether the workspace
|
||||
wants commands confirmed. Those rules lived inline in the widget, where no test
|
||||
could reach them — and each fails silently when wrong (a duplicated user message,
|
||||
or a command that quietly stops asking for approval).
|
||||
|
||||
They live here instead, as the mapping step the migration map assigns to the
|
||||
application layer. The widget keeps only what is genuinely widget-specific:
|
||||
reading its own state and building the provider.
|
||||
|
||||
Layer rules (``docs/architecture/ADR-001-layered-architecture.md``): pure Python.
|
||||
Everything arrives as a plain value, so this module never sees a widget.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, Optional, Sequence
|
||||
|
||||
from ...domain.agents.conversation_execution_request import ConversationExecutionRequest
|
||||
|
||||
|
||||
def build_cowork_turn_request(
|
||||
*,
|
||||
turn_id: str,
|
||||
session_id: str,
|
||||
messages: Sequence[Dict[str, Any]],
|
||||
surface: str = "cowork",
|
||||
project_id: str = "",
|
||||
title: str = "",
|
||||
provider_id: str = "",
|
||||
model: str = "",
|
||||
instructions: str = "",
|
||||
output_dir: Optional[Any] = None,
|
||||
home_output_root: Optional[Any] = None,
|
||||
confirm_commands: bool = False,
|
||||
agent_role: str = "cowork",
|
||||
) -> ConversationExecutionRequest:
|
||||
"""Build one Cowork turn's immutable request.
|
||||
|
||||
``messages`` is the widget's working list, which ALREADY ends with this
|
||||
turn's user message (the chat panel composes it — prefix, attachments,
|
||||
session notes — before the job starts). So the prompt is that last message
|
||||
and the history is everything before it. The request records both; the
|
||||
service is handed the same working list and appends into it.
|
||||
|
||||
Keyword-only on purpose: a dozen positional strings in a call site is exactly
|
||||
how a title ends up in the project-id slot.
|
||||
"""
|
||||
history = list(messages or ())
|
||||
# ``pop`` rather than ``[-1]``/``[:-1]`` so the empty-list case needs no
|
||||
# special branch: a turn with nothing in it yields an empty prompt instead of
|
||||
# raising IndexError deep inside a worker thread.
|
||||
last = history.pop() if history else {}
|
||||
return ConversationExecutionRequest(
|
||||
turn_id=turn_id,
|
||||
session_id=session_id,
|
||||
surface=surface,
|
||||
project_id=project_id,
|
||||
title=title,
|
||||
prompt=str(last.get("content") or ""),
|
||||
messages=history,
|
||||
provider_id=provider_id,
|
||||
model=model,
|
||||
project_context=instructions,
|
||||
output_dir=output_dir,
|
||||
home_output_root=home_output_root,
|
||||
# The workspace's Auto-run override (or the global setting) decides
|
||||
# whether run_command/install_package must be approved first.
|
||||
gate_mode="confirm" if confirm_commands else "auto",
|
||||
agent_role=agent_role,
|
||||
)
|
||||
|
||||
|
||||
__all__ = ["build_cowork_turn_request"]
|
||||
@@ -0,0 +1,177 @@
|
||||
"""The seams :mod:`conversation_application_service` runs a turn through (R04-T03).
|
||||
|
||||
Two Protocols and six callables — chosen deliberately, not by reflex. The
|
||||
refactor plan forbids giving every class an interface, so a contract exists here
|
||||
only where there is both a real ``core/*`` implementation AND a test double:
|
||||
|
||||
* :class:`ModelCallPort` — one provider round-trip *including* the app's
|
||||
existing context-overflow recovery, which is why the raw ``Provider.chat``
|
||||
signature is not enough.
|
||||
* :class:`ToolRuntimePort` — the tool + output-folder runtime, kept as one
|
||||
cohesive object because every method operates on the same sandbox.
|
||||
|
||||
Everything else is a single function, so it is expressed as a callable type
|
||||
rather than a class with one method (the same choice R03 made for
|
||||
``ConfirmationCallback``). All of them are optional: a service built with none
|
||||
of them still runs a plain chat turn, which is what keeps the unit tests short.
|
||||
|
||||
Layer rules (``docs/architecture/ADR-001-layered-architecture.md``): application
|
||||
layer — pure Python. Nothing here imports PySide6, ``core.*``, ``providers.*``
|
||||
or ``ui.*``; the concrete wiring lives in :mod:`core_runtime_adapter`.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import (
|
||||
Any,
|
||||
Callable,
|
||||
Dict,
|
||||
List,
|
||||
Optional,
|
||||
Protocol,
|
||||
Sequence,
|
||||
Tuple,
|
||||
runtime_checkable,
|
||||
)
|
||||
|
||||
from ...domain.agents.agent_event import AgentEvent, ToolPreview
|
||||
|
||||
# The plan tool is special-cased by the loop: it drives the Plan panel and
|
||||
# produces no chat bubble and no file. Named here so the check is not a bare
|
||||
# string literal in the middle of the dispatch.
|
||||
PLAN_TOOL = "update_plan"
|
||||
|
||||
# Tools that need approval before they run when the workspace is in confirm
|
||||
# mode. R05 replaces this tuple with a real ``ToolPolicyGateway`` keyed on
|
||||
# ToolCapability; until then it mirrors exactly what the runtime gates today.
|
||||
GATED_TOOLS = ("run_command", "install_package")
|
||||
|
||||
# Shown when the user (or the workspace policy) rejects a proposed command. The
|
||||
# exact string also becomes the tool message the model reads back, so it must
|
||||
# stay stable.
|
||||
REJECTED_OUTPUT = "Rejected by user."
|
||||
|
||||
# A reasoning model can answer with thinking only. The note is written into the
|
||||
# assistant message itself, not merely emitted, so an unattended run does not
|
||||
# read back an empty answer and report "(no output)".
|
||||
REASONING_ONLY_NOTE = "*(model returned only its reasoning — try rephrasing)*"
|
||||
|
||||
# Emitted when the turn is stopped by its own safety ceiling rather than by the
|
||||
# model finishing. Never silent: being cut off looks exactly like being done.
|
||||
BUDGET_NOTE_TEMPLATE = (
|
||||
"\n\n⚠️ Reached the {steps}-step safety limit before the task signalled "
|
||||
"completion — stopping here. Re-run to continue if more work remains."
|
||||
)
|
||||
|
||||
|
||||
def combine_instructions(*blocks: Optional[str]) -> str:
|
||||
"""Join the standing-instruction blocks of a turn, skipping the absent ones.
|
||||
|
||||
A turn's instructions arrive as several independent blocks — the project's
|
||||
shared context, an Admin agent's persona, a skill's rules, the
|
||||
"this runs unattended" reminder — and each caller was joining them inline
|
||||
with its own ``f"{a}\\n\\n{b}" if a else b`` expression. Two call sites now
|
||||
need the same rule (the Cowork widget in R04-T04 and the task runner in
|
||||
R04-T05), which is the point at which it stops being an expression.
|
||||
|
||||
Whitespace-only blocks count as absent: they would otherwise open the system
|
||||
prompt with a stray blank line.
|
||||
"""
|
||||
return "\n\n".join(b.strip() for b in blocks if b and b.strip())
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Callables.
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Receives every typed event the turn produces. The caller decides what that
|
||||
# means — render it, forward it as a legacy dict, autosave on it.
|
||||
EventSink = Callable[[AgentEvent], None]
|
||||
|
||||
# True once the user has asked to stop. Polled between steps and between tool
|
||||
# calls, the same cadence the current runtime uses.
|
||||
CancelFn = Callable[[], bool]
|
||||
|
||||
# ``(prompt, attachment_paths) -> body``. Runs on the worker thread because
|
||||
# extracting a .docx may pip-install a parser or call LibreOffice.
|
||||
AttachmentReader = Callable[[str, Tuple[str, ...]], str]
|
||||
|
||||
# ``(messages, advertised_tool_names) -> None`` — inserts the system prompt and
|
||||
# folds in skills, security rules and project instructions, in place. It needs
|
||||
# the tool names because the system prompt gains an MS365 paragraph only when
|
||||
# ms365 tools are actually present.
|
||||
PromptPreparer = Callable[[List[Dict[str, Any]], Tuple[str, ...]], None]
|
||||
|
||||
# Reviews the assembled request; raises to refuse the turn outright.
|
||||
PromptGuard = Callable[[List[Dict[str, Any]]], None]
|
||||
|
||||
# Reviews one proposed tool call; raises to refuse it.
|
||||
CommandGuard = Callable[[str, Dict[str, Any]], None]
|
||||
|
||||
# ``(messages, cancel) -> None``. Summarises old turns in place when the
|
||||
# conversation nears the model's context budget; a no-op when compaction is off
|
||||
# or the conversation is short. It takes the cancel signal because compacting
|
||||
# calls the model itself, so Stop has to reach it too.
|
||||
ContextCompactor = Callable[[List[Dict[str, Any]], "CancelFn"], None]
|
||||
|
||||
# ``(action) -> approved``. Blocks the worker thread while a human decides.
|
||||
PermissionRequest = Callable[[Dict[str, Any]], bool]
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Ports.
|
||||
# --------------------------------------------------------------------------- #
|
||||
@runtime_checkable
|
||||
class ModelCallPort(Protocol):
|
||||
"""One call to the model, with the app's retry/recovery behaviour applied."""
|
||||
|
||||
def call(self, messages: List[Dict[str, Any]], tools: Sequence[Any],
|
||||
on_text: Optional[Callable[[str], None]] = None,
|
||||
on_reasoning: Optional[Callable[[str], None]] = None,
|
||||
cancel: Optional[CancelFn] = None) -> Dict[str, Any]:
|
||||
"""Return the canonical assistant message (content plus tool calls)."""
|
||||
|
||||
|
||||
@runtime_checkable
|
||||
class ToolRuntimePort(Protocol):
|
||||
"""The tools a turn may call, and the folder its files land in."""
|
||||
|
||||
def specs(self, allowed_tools: Optional[Sequence[str]] = None) -> Sequence[Any]:
|
||||
"""Tool specs to advertise to the model, already filtered.
|
||||
|
||||
Returns opaque objects (the provider layer's ``ToolSpec``); the service
|
||||
only ever reads ``.name`` off them, which is what keeps this layer free
|
||||
of a provider import.
|
||||
"""
|
||||
|
||||
def preview(self, name: str, args: Dict[str, Any]) -> Optional[ToolPreview]:
|
||||
"""Human-readable description of a call that is about to run."""
|
||||
|
||||
def execute(self, name: str, args: Dict[str, Any],
|
||||
on_output: Optional[Callable[[str], None]] = None,
|
||||
cancel: Optional[CancelFn] = None) -> Dict[str, Any]:
|
||||
"""Run one tool call.
|
||||
|
||||
Returns the runtime's own result mapping: ``ok``, ``output``, optionally
|
||||
``path``/``produced`` for files it created, and ``plan_steps`` for the
|
||||
plan tool.
|
||||
"""
|
||||
|
||||
def snapshot(self) -> Any:
|
||||
"""Opaque record of the output folder before the turn started."""
|
||||
|
||||
def finalize(self, before: Any, cancelled: bool = False
|
||||
) -> Tuple[Sequence[str], Sequence[str]]:
|
||||
"""Tidy the output folder; return ``(removed_paths, added_paths)``.
|
||||
|
||||
Not read-only — it deletes the scratch sandbox and flattens sub-folders —
|
||||
so the service only calls it for a turn that actually started.
|
||||
"""
|
||||
|
||||
|
||||
__all__ = [
|
||||
"PLAN_TOOL", "GATED_TOOLS", "REJECTED_OUTPUT", "REASONING_ONLY_NOTE",
|
||||
"BUDGET_NOTE_TEMPLATE", "combine_instructions",
|
||||
"EventSink", "CancelFn", "AttachmentReader", "PromptPreparer", "PromptGuard",
|
||||
"CommandGuard", "ContextCompactor", "PermissionRequest",
|
||||
"ModelCallPort", "ToolRuntimePort",
|
||||
]
|
||||
Reference in New Issue
Block a user