Compare commits
15
Commits
@@ -39,3 +39,32 @@ jobs:
|
||||
|
||||
- name: Run tests
|
||||
run: python -m pytest tests -q
|
||||
|
||||
# --- CASAN Verification Gate -------------------------------------
|
||||
# Ba check này là điều kiện của cổng ngày 30/08. Chạy trên MỌI PR để
|
||||
# biết vi phạm ngay hôm phát sinh, thay vì dồn tới ngày cổng.
|
||||
#
|
||||
# Check 1 do Team Gamma sở hữu và đã có. Check 2 (Team Hoa) và Check 3
|
||||
# (Team Duy) chưa viết — bước dưới bỏ qua nếu script chưa tồn tại, để
|
||||
# thêm cổng không làm đỏ CI của hai team kia.
|
||||
|
||||
- name: "CASAN Check 1 — không có credential lộ (Team Gamma)"
|
||||
run: |
|
||||
python scripts/audit_security.py --self-test
|
||||
python scripts/audit_security.py
|
||||
|
||||
- name: "CASAN Check 2 — file production ≤ 400 dòng (Team Hoa)"
|
||||
run: |
|
||||
if [ -f scripts/check_loc.py ]; then
|
||||
python scripts/check_loc.py
|
||||
else
|
||||
echo "scripts/check_loc.py chưa có — Team Hoa viết, hạn 30/08. Bỏ qua."
|
||||
fi
|
||||
|
||||
- name: "CASAN Check 3 — domain/ và application/ không import PySide6 (Team Duy)"
|
||||
run: |
|
||||
if [ -f scripts/check_imports.py ]; then
|
||||
python scripts/check_imports.py
|
||||
else
|
||||
echo "scripts/check_imports.py chưa có — Team Duy viết, hạn 30/08. Bỏ qua."
|
||||
fi
|
||||
|
||||
+11
-5
@@ -28,7 +28,10 @@ bower_components/
|
||||
.env.preview
|
||||
*.pem
|
||||
*.key
|
||||
secrets/
|
||||
# Neo vào gốc repo: mẫu không neo nuốt MỌI thư mục tên secrets ở mọi độ
|
||||
# sâu — nó đã âm thầm chặn infrastructure/secrets/ (mã nguồn, không phải
|
||||
# bí mật) khỏi repo suốt 21-22/08.
|
||||
/secrets/
|
||||
credentials.json
|
||||
.npmrc
|
||||
.yarnrc
|
||||
@@ -36,9 +39,11 @@ credentials.json
|
||||
# =============================================================================
|
||||
# Build & Distribution
|
||||
# =============================================================================
|
||||
dist/
|
||||
build/
|
||||
out/
|
||||
# Neo vao goc — mau khong neo se nuot moi thu muc trung ten o moi do sau,
|
||||
# ke ca ma nguon. Da mac dung loi do voi secrets/ (xem khoi Credentials).
|
||||
/dist/
|
||||
/build/
|
||||
/out/
|
||||
.next/
|
||||
.nuxt/
|
||||
.output/
|
||||
@@ -73,7 +78,8 @@ desktop.ini
|
||||
# Logs & Debug
|
||||
# =============================================================================
|
||||
*.log
|
||||
logs/
|
||||
# Neo vao goc: infrastructure/logs/ la ma nguon, khong phai log chay may.
|
||||
/logs/
|
||||
npm-debug.log*
|
||||
yarn-debug.log*
|
||||
yarn-error.log*
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
"""adapters/ — Adapter riêng cho Qt (clock, thread, timer).
|
||||
|
||||
Kế hoạch gốc đặt tên thư mục này là ``platform/``. Không dùng được: chạy
|
||||
bất kỳ script nào từ thư mục gốc repo (``python tools/...``,
|
||||
``python scripts/...``) thì ``platform/`` **che khuất module ``platform``
|
||||
của thư viện chuẩn**, và ``import keyring`` chết ngay với
|
||||
``AttributeError: module 'platform' has no attribute 'system'``.
|
||||
Repo có 26 script chạy đúng kiểu đó.
|
||||
|
||||
Đổi tên là cách duy nhất chắc chắn — không thể bắt mọi người nhớ "đừng bao
|
||||
giờ chạy python từ thư mục gốc".
|
||||
"""
|
||||
+1
-12
@@ -1,12 +1 @@
|
||||
"""Application layer - pure Python use-case orchestration.
|
||||
|
||||
Sits between ``presentation/`` (Qt widgets) and ``domain/`` (entities). A module
|
||||
here answers "what has to happen, in what order" for one use case - route a
|
||||
turn, run a conversation - without knowing whether a human, a scheduler or a
|
||||
test triggered it.
|
||||
|
||||
Hard rule (ADR-001 I1/I3, enforced by ``scripts/check_imports.py``): no
|
||||
PySide6/PyQt imports and no reach into ``presentation/``/``ui/``. Results travel
|
||||
back up through plain-Python callbacks; turning those into Qt signals is the
|
||||
presentation layer's job.
|
||||
"""
|
||||
"""application/ — Điều phối use-case. KHÔNG import PySide6. Gọi domain + interface hạ tầng."""
|
||||
|
||||
@@ -1,8 +0,0 @@
|
||||
"""Conversation use case: the lifecycle of one agent turn (EPIC R04)."""
|
||||
|
||||
from .conversation_application_service import (
|
||||
ConversationApplicationService,
|
||||
TurnResult,
|
||||
)
|
||||
|
||||
__all__ = ["ConversationApplicationService", "TurnResult"]
|
||||
|
||||
@@ -1,328 +0,0 @@
|
||||
"""ConversationApplicationService - the turn lifecycle, outside the widget (R04-T03).
|
||||
|
||||
What this replaces
|
||||
------------------
|
||||
The lifecycle of one Cowork turn is currently spread across a closure inside
|
||||
``ui/cowork_tab.py::build_job`` and a second, near-identical assembly inside
|
||||
``core/task_executors.py::_run_agent``. Both:
|
||||
|
||||
* read live UI/config state from a worker thread,
|
||||
* build the provider, the MCP tool set and the project context by hand,
|
||||
* call ``core.chat_agent.run_cowork`` with a dozen positional-ish arguments,
|
||||
* consume untyped event dicts.
|
||||
|
||||
Two copies means a fix to one path (say, promoting output files on failure)
|
||||
silently misses the other. This service is the single implementation: it takes
|
||||
an immutable :class:`ConversationExecutionRequest`, runs the turn, and reports
|
||||
typed :class:`AgentEvent` objects.
|
||||
|
||||
What it deliberately does NOT do
|
||||
--------------------------------
|
||||
It does not re-implement the agent loop. ``run_cowork`` stays the engine
|
||||
(strangler fig, ADR-001 section 4) and keeps its characterization tests
|
||||
(``tests/characterization/test_run_cowork.py``). This layer owns the parts that
|
||||
were tangled into the UI: assembling the call, translating events, and giving a
|
||||
turn a well-defined end.
|
||||
|
||||
Pure Python: no Qt import, no config access. Everything it needs arrives through
|
||||
constructor callbacks, so the same service runs a turn from a chat panel, from
|
||||
the scheduler, or from a test.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable, Dict, List, Optional, Tuple
|
||||
|
||||
from cowork_local.domain.agents.agent_event import (
|
||||
AgentEvent,
|
||||
ErrorEvent,
|
||||
TurnCompletedEvent,
|
||||
collect_text,
|
||||
event_from_dict,
|
||||
)
|
||||
from cowork_local.domain.agents.conversation_execution_request import (
|
||||
ConversationExecutionRequest,
|
||||
)
|
||||
|
||||
logger = logging.getLogger("cowork_local.conversations")
|
||||
|
||||
# Presentation/scheduler supplies these. Kept as plain callables (not objects)
|
||||
# so a test can wire the service with three lambdas.
|
||||
EventCallback = Callable[[AgentEvent], None]
|
||||
CancelFn = Callable[[], bool]
|
||||
ProviderFactory = Callable[[str, str], Any] # (provider_id, model) -> Provider
|
||||
ToolSourceFactory = Callable[[], Tuple[Any, Any]] # () -> (extra_tools, extra_executor)
|
||||
GateFactory = Callable[[ConversationExecutionRequest], Any] # -> PermissionGate or None
|
||||
|
||||
|
||||
@dataclass
|
||||
class TurnResult:
|
||||
"""What a finished turn produced.
|
||||
|
||||
``messages`` is the conversation AFTER the turn (system prompt inserted,
|
||||
assistant and tool messages appended) - the caller persists this as the new
|
||||
history. ``final_text`` is the visible answer, reasoning excluded.
|
||||
"""
|
||||
|
||||
request: ConversationExecutionRequest
|
||||
messages: List[Dict[str, Any]] = field(default_factory=list)
|
||||
events: List[AgentEvent] = field(default_factory=list)
|
||||
final_text: str = ""
|
||||
cancelled: bool = False
|
||||
error: str = ""
|
||||
# The original exception, kept alongside its message so a caller that needs
|
||||
# to preserve legacy failure handling can re-raise the SAME object rather
|
||||
# than a lookalike (SecurityBlocked, for instance, carries context that a
|
||||
# re-wrapped RuntimeError would lose).
|
||||
exception: Optional[BaseException] = None
|
||||
|
||||
@property
|
||||
def ok(self) -> bool:
|
||||
"""True when the turn completed without an error and without a Stop."""
|
||||
return not self.error and not self.cancelled
|
||||
|
||||
def raise_if_failed(self) -> None:
|
||||
"""Re-raise the turn's failure, if any.
|
||||
|
||||
Callers that already have failure handling built around an exception
|
||||
(the Qt worker turns one into its ``failed`` signal) use this to keep
|
||||
that path intact while still getting a TurnResult on success."""
|
||||
if self.exception is not None:
|
||||
raise self.exception
|
||||
|
||||
def output_dir(self) -> Optional[Path]:
|
||||
"""This turn's output folder, or None when it could not write files."""
|
||||
return Path(self.request.output_dir) if self.request.output_dir else None
|
||||
|
||||
|
||||
class ConversationApplicationService:
|
||||
"""Runs one agent turn from an immutable request.
|
||||
|
||||
Args:
|
||||
provider_factory: ``(provider_id, model) -> Provider``. Production passes
|
||||
``AppContext.build_provider_for``; tests pass a lambda returning a
|
||||
:class:`FakeProvider`.
|
||||
tool_source: ``() -> (extra_tools, extra_executor)`` for MCP/connector
|
||||
tools. Optional - a turn with no external tools passes nothing.
|
||||
gate_factory: ``(request) -> PermissionGate | None``, consulted when the
|
||||
request asks to confirm commands. Optional for the same reason.
|
||||
runner: the turn engine. Defaults to ``core.chat_agent.run_cowork``,
|
||||
imported lazily so this module stays importable (and testable)
|
||||
without pulling in the whole legacy tool stack.
|
||||
security_config: the app config the security layers read. ``None``
|
||||
disables them, which is what headless callers already rely on.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
provider_factory: ProviderFactory,
|
||||
*,
|
||||
tool_source: Optional[ToolSourceFactory] = None,
|
||||
gate_factory: Optional[GateFactory] = None,
|
||||
runner: Optional[Callable[..., Any]] = None,
|
||||
security_config: Any = None,
|
||||
) -> None:
|
||||
self._provider_factory = provider_factory
|
||||
self._tool_source = tool_source
|
||||
self._gate_factory = gate_factory
|
||||
self._runner = runner
|
||||
self._security_config = security_config
|
||||
|
||||
# -- main entry point -------------------------------------------------- #
|
||||
def run_turn(
|
||||
self,
|
||||
request: ConversationExecutionRequest,
|
||||
on_event: Optional[EventCallback] = None,
|
||||
cancel: Optional[CancelFn] = None,
|
||||
) -> TurnResult:
|
||||
"""Execute one turn and return everything it produced.
|
||||
|
||||
Never raises: a provider or tool failure becomes an :class:`ErrorEvent`
|
||||
plus ``TurnResult.error``. Callers run this on a worker thread and have
|
||||
no good way to handle an exception crossing that boundary - today an
|
||||
escaped error kills the worker and the UI just stops updating, with no
|
||||
message shown.
|
||||
|
||||
Exactly one :class:`TurnCompletedEvent` is always emitted last, whether
|
||||
the turn succeeded, failed or was cancelled. That is the end-of-turn
|
||||
signal the legacy engine never had.
|
||||
"""
|
||||
return self.execute_turn(self.begin_turn(request), on_event=on_event, cancel=cancel)
|
||||
|
||||
def begin_turn(self, request: ConversationExecutionRequest) -> TurnResult:
|
||||
"""Create the (still empty) result a turn will fill in.
|
||||
|
||||
Exposed separately from :meth:`run_turn` because some callers need the
|
||||
LIVE message list while the turn is running, not only afterwards: the
|
||||
scheduler re-saves the conversation to History after every assistant
|
||||
message so a long unattended run shows live progress when reopened.
|
||||
Handing them ``result.messages`` - the very list the engine appends to -
|
||||
is what makes that possible without leaking the engine into the caller.
|
||||
"""
|
||||
return TurnResult(request=request, messages=request.message_list())
|
||||
|
||||
def execute_turn(
|
||||
self,
|
||||
result: TurnResult,
|
||||
on_event: Optional[EventCallback] = None,
|
||||
cancel: Optional[CancelFn] = None,
|
||||
) -> TurnResult:
|
||||
"""Run a turn previously created by :meth:`begin_turn`. See
|
||||
:meth:`run_turn` for the error/cancellation contract."""
|
||||
request = result.request
|
||||
emit = self._make_emitter(result, on_event)
|
||||
cancel = cancel or (lambda: False)
|
||||
|
||||
try:
|
||||
self._execute(request, result, emit, cancel)
|
||||
except Exception as exc: # noqa: BLE001 - see docstring
|
||||
result.error = str(exc) or exc.__class__.__name__
|
||||
result.exception = exc
|
||||
logger.exception("turn %s failed", request.turn_id)
|
||||
emit(ErrorEvent(message=result.error,
|
||||
recoverable=self._is_recoverable(exc)))
|
||||
|
||||
result.cancelled = bool(cancel())
|
||||
result.final_text = collect_text(result.events) or self._last_assistant_text(result.messages)
|
||||
emit(TurnCompletedEvent(content=result.final_text, cancelled=result.cancelled))
|
||||
return result
|
||||
|
||||
# -- internals --------------------------------------------------------- #
|
||||
def _execute(self, request: ConversationExecutionRequest, result: TurnResult,
|
||||
emit: Callable[[AgentEvent], None], cancel: CancelFn) -> None:
|
||||
"""Assemble the engine call from the request snapshot and run it."""
|
||||
provider = self._provider_factory(request.provider, request.model)
|
||||
extra_tools, extra_executor = self._resolve_tools()
|
||||
gate = self._resolve_gate(request)
|
||||
|
||||
# The engine speaks untyped dicts; bridge them into typed events at this
|
||||
# single point rather than at every consumer.
|
||||
def legacy_emit(payload: Dict[str, Any]) -> None:
|
||||
event = event_from_dict(payload)
|
||||
if event is not None:
|
||||
emit(event)
|
||||
|
||||
run = self._resolve_runner()
|
||||
run(
|
||||
provider,
|
||||
result.messages, # mutated in place by the engine, as before
|
||||
self._output_dir(request),
|
||||
legacy_emit,
|
||||
cancel,
|
||||
title=request.title,
|
||||
extra_tools=extra_tools,
|
||||
extra_executor=extra_executor,
|
||||
project_context=request.project_context,
|
||||
security_config=self._security_config,
|
||||
gate=gate,
|
||||
allowed_tools=list(request.allowed_tools) if request.allowed_tools is not None else None,
|
||||
max_steps=request.max_steps,
|
||||
run_to_completion=request.run_to_completion,
|
||||
completion_max_steps=request.completion_max_steps,
|
||||
enforce_rules=request.enforce_rules,
|
||||
**self._role_kwargs(request),
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _make_emitter(result: TurnResult,
|
||||
on_event: Optional[EventCallback]) -> Callable[[AgentEvent], None]:
|
||||
"""Record every event on the result AND forward it to the caller.
|
||||
|
||||
Recording is unconditional so a headless caller (the scheduler) can read
|
||||
the full event list afterwards without having to supply a callback just
|
||||
to collect it - which is exactly what task_executors does today with an
|
||||
ad-hoc list.
|
||||
"""
|
||||
def emit(event: AgentEvent) -> None:
|
||||
result.events.append(event)
|
||||
if on_event is None:
|
||||
return
|
||||
try:
|
||||
on_event(event)
|
||||
except Exception: # noqa: BLE001
|
||||
# A consumer that throws (a closing widget, say) must not abort
|
||||
# the turn that is feeding it.
|
||||
logger.debug("event consumer raised for %s", event.type, exc_info=True)
|
||||
return emit
|
||||
|
||||
def _resolve_runner(self) -> Callable[..., Any]:
|
||||
"""The turn engine, imported lazily on first use."""
|
||||
if self._runner is None:
|
||||
from cowork_local.core.chat_agent import run_cowork
|
||||
|
||||
self._runner = run_cowork
|
||||
return self._runner
|
||||
|
||||
def _resolve_tools(self) -> Tuple[Any, Any]:
|
||||
"""MCP/connector tools for this turn, or ``(None, None)``.
|
||||
|
||||
A failure here degrades to "no external tools" rather than failing the
|
||||
turn: an MCP server that will not start must not stop the user from
|
||||
chatting, which is the behaviour the chat panel already relies on.
|
||||
"""
|
||||
if self._tool_source is None:
|
||||
return None, None
|
||||
try:
|
||||
return self._tool_source()
|
||||
except Exception: # noqa: BLE001
|
||||
logger.warning("tool source unavailable - running without external tools",
|
||||
exc_info=True)
|
||||
return None, None
|
||||
|
||||
def _resolve_gate(self, request: ConversationExecutionRequest) -> Any:
|
||||
"""The permission gate, when this turn asked to confirm commands."""
|
||||
if not request.confirm_commands or self._gate_factory is None:
|
||||
return None
|
||||
return self._gate_factory(request)
|
||||
|
||||
@staticmethod
|
||||
def _output_dir(request: ConversationExecutionRequest) -> Path:
|
||||
"""The turn's output folder as a Path.
|
||||
|
||||
The request holds it as a string to stay serialisable; converting at the
|
||||
single point of use keeps that decision from leaking into every caller.
|
||||
"""
|
||||
return Path(request.output_dir) if request.output_dir else Path.cwd()
|
||||
|
||||
@staticmethod
|
||||
def _role_kwargs(request: ConversationExecutionRequest) -> Dict[str, Any]:
|
||||
"""``agent_role`` only when the request set one.
|
||||
|
||||
Omitted otherwise so the engine applies its own default (the interactive
|
||||
Cowork role) instead of being handed an empty string, which would land
|
||||
in the audit log as an unattributed tool call.
|
||||
"""
|
||||
return {"agent_role": request.agent_role} if request.agent_role else {}
|
||||
|
||||
@staticmethod
|
||||
def _last_assistant_text(messages: List[Dict[str, Any]]) -> str:
|
||||
"""Fallback answer text when no text events were seen.
|
||||
|
||||
A turn whose whole answer arrived in one non-streamed message still has
|
||||
to report a final answer - the scheduler writes it into output.md, and
|
||||
an empty string there reads as "(no output)".
|
||||
"""
|
||||
for message in reversed(messages):
|
||||
if message.get("role") == "assistant" and (message.get("content") or "").strip():
|
||||
return str(message["content"])
|
||||
return ""
|
||||
|
||||
@staticmethod
|
||||
def _is_recoverable(exc: Exception) -> bool:
|
||||
"""Whether the user can act on this failure themselves.
|
||||
|
||||
"Model not found" is the motivating case: the chat panel restores the
|
||||
typed message into the composer so the user can switch model and resend
|
||||
instead of retyping it (see providers/base.py::MODEL_NOT_FOUND_HINT).
|
||||
"""
|
||||
try:
|
||||
from cowork_local.providers.base import is_model_not_found_error
|
||||
|
||||
return bool(is_model_not_found_error(str(exc)))
|
||||
except Exception: # noqa: BLE001
|
||||
return False
|
||||
|
||||
|
||||
__all__ = ["ConversationApplicationService", "TurnResult"]
|
||||
@@ -1,12 +0,0 @@
|
||||
"""Model routing use case: pick the best-fit model for one turn (EPIC R03)."""
|
||||
|
||||
from .routing_application_service import (
|
||||
RoutingApplicationService,
|
||||
RoutingDecision,
|
||||
RoutingMode,
|
||||
is_valid_mode,
|
||||
normalize_mode,
|
||||
)
|
||||
|
||||
__all__ = ["RoutingApplicationService", "RoutingDecision", "RoutingMode",
|
||||
"normalize_mode", "is_valid_mode"]
|
||||
|
||||
@@ -1,353 +0,0 @@
|
||||
"""RoutingApplicationService - one routing flow for every surface (R03-T03).
|
||||
|
||||
Before this service, the same routing algorithm existed three times:
|
||||
|
||||
* ``ui/chat_panel.py::_apply_routing`` (Cowork chat)
|
||||
* ``ui/co4e_tab.py::_apply_co4e_routing`` (Co4E studio)
|
||||
* ``ui/folder_tab.py::_ai_apply_routing`` (AI-Edit)
|
||||
|
||||
The three copies had already drifted - each one resolves the "current model"
|
||||
differently and each one has its own private notion of what to do when the user
|
||||
declines - and every one of them lives inside a Qt widget, so none of the logic
|
||||
could be tested without building a window.
|
||||
|
||||
This module is the single implementation. It is pure Python: no Qt import, no
|
||||
config access, no network. The presentation layer supplies a confirm callback
|
||||
and renders the notice; everything else happens here.
|
||||
|
||||
Modes (:class:`RoutingMode`)
|
||||
----------------------------
|
||||
* ``OFF`` - never switch. The user's pinned model always wins.
|
||||
* ``AUTO`` - switch silently when the best candidate clears the gain threshold.
|
||||
* ``MANUAL`` - propose the switch and switch only if the confirm callback approves.
|
||||
* ``FALLBACK`` - never switch pre-emptively; switch only AFTER the current model
|
||||
fails, to the next-best candidate. This is the mode a user wants when they
|
||||
trust their own model choice but still want the turn to survive an outage.
|
||||
|
||||
Migration note (ADR-001 section 4): the scoring/ranking engine is NOT rewritten.
|
||||
This service depends on the small :class:`RoutingPort` interface, and production
|
||||
wires the existing, already-tested ``core.routing.service.RoutingService`` into
|
||||
it. Tests wire a fake.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from enum import Enum
|
||||
from typing import Any, Callable, List, Optional, Protocol, Sequence, Tuple
|
||||
|
||||
|
||||
class RoutingMode(str, Enum):
|
||||
"""Per-surface routing behaviour.
|
||||
|
||||
The first three values match ``core.routing.models.SwitchMode`` string for
|
||||
string, so a mode read from the existing config round-trips unchanged.
|
||||
"""
|
||||
|
||||
OFF = "off"
|
||||
AUTO = "auto"
|
||||
MANUAL = "manual"
|
||||
FALLBACK = "fallback"
|
||||
|
||||
@classmethod
|
||||
def parse(cls, raw: Any) -> "RoutingMode":
|
||||
"""Best-effort parse of a config value.
|
||||
|
||||
Unknown or empty values become ``OFF``: routing is an optimisation, and
|
||||
the safe reading of a corrupt setting is "leave the user's model alone"
|
||||
rather than "silently move their work to another model".
|
||||
"""
|
||||
try:
|
||||
return cls(str(raw or "off").strip().lower())
|
||||
except ValueError:
|
||||
return cls.OFF
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RoutingDecision:
|
||||
"""The outcome of routing one turn - an immutable instruction for the caller.
|
||||
|
||||
``provider``/``model`` are ALWAYS filled with what the turn should actually
|
||||
run on, switched or not, so a call site never has to re-derive the fallback
|
||||
itself (the bug that made the three UI copies diverge).
|
||||
"""
|
||||
|
||||
mode: RoutingMode
|
||||
provider: str
|
||||
model: str
|
||||
switched: bool = False
|
||||
task_type: str = ""
|
||||
score_gain: float = 0.0
|
||||
reason: str = ""
|
||||
declined: bool = False # Manual mode: a switch was offered and refused
|
||||
# What the turn would have run on without routing. Carried so the Manual
|
||||
# confirm dialog can show "from X to Y" without re-deriving the current
|
||||
# model itself - re-deriving it differently per screen is exactly how the
|
||||
# three legacy copies drifted apart.
|
||||
previous_provider: str = ""
|
||||
previous_model: str = ""
|
||||
|
||||
@property
|
||||
def should_notify(self) -> bool:
|
||||
"""True when the UI should show the "switched model" notice - i.e. only
|
||||
when a switch really happened."""
|
||||
return self.switched
|
||||
|
||||
def target(self) -> Tuple[str, str]:
|
||||
"""``(provider, model)`` to run this turn on."""
|
||||
return self.provider, self.model
|
||||
|
||||
@property
|
||||
def from_model(self) -> str:
|
||||
"""Candidate key (``provider/model``) of the model being switched away
|
||||
from, or "" when nothing was selected yet.
|
||||
|
||||
Named to match ``core.routing.models.SwitchDecision`` so the existing
|
||||
Manual-mode dialog (``ui/routing_toggle.py::confirm_switch``) accepts
|
||||
this object unchanged - the dialog moves to the new shape in EPIC R08.
|
||||
"""
|
||||
if not self.previous_model:
|
||||
return ""
|
||||
return f"{self.previous_provider}/{self.previous_model}"
|
||||
|
||||
@property
|
||||
def to_model(self) -> str:
|
||||
"""Candidate key (``provider/model``) of the model to run on. See
|
||||
:attr:`from_model` for why the name matches the legacy decision."""
|
||||
return f"{self.provider}/{self.model}" if self.model else ""
|
||||
|
||||
|
||||
def is_valid_mode(raw: Any) -> bool:
|
||||
"""True when ``raw`` names a mode the routing service understands.
|
||||
|
||||
Distinct from :func:`normalize_mode` because callers need to tell "the user
|
||||
chose off" apart from "this stored value is unrecognised" - the per-workspace
|
||||
lookup falls back to the global setting only in the second case.
|
||||
"""
|
||||
try:
|
||||
RoutingMode(str(raw or "").strip().lower())
|
||||
except ValueError:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def normalize_mode(raw: Any) -> str:
|
||||
"""Canonical mode string for persistence, or ``"off"`` when unrecognised.
|
||||
|
||||
Exists so the mode vocabulary is defined exactly once. It used to be
|
||||
hard-coded as a ``("off", "auto", "manual")`` tuple in four separate places
|
||||
(config.py twice, state.py twice); adding FALLBACK meant finding all four,
|
||||
and missing one silently downgraded the user's choice back to "off".
|
||||
"""
|
||||
return RoutingMode.parse(raw).value
|
||||
|
||||
|
||||
class RoutingPort(Protocol):
|
||||
"""The slice of the routing engine this service needs.
|
||||
|
||||
Declared as a Protocol so the application layer states its requirement
|
||||
without importing the implementation - which is what lets the whole service
|
||||
be tested against a 20-line fake, and lets ``core.routing`` be replaced later
|
||||
without touching this file.
|
||||
"""
|
||||
|
||||
def route(self, surface: str, prompt: str, current_provider: str, current_model: str,
|
||||
*, mode_override: Optional[str] = None,
|
||||
required_capabilities: Optional[List[str]] = None,
|
||||
task_type: Optional[Any] = None) -> Any:
|
||||
"""Return a route result exposing ``should_switch``, ``target()``,
|
||||
``task_type`` and ``decision``."""
|
||||
|
||||
|
||||
# Presentation supplies this to ask the human. Receives the proposal so the
|
||||
# dialog can explain it; returns True to approve. Manual mode only.
|
||||
ConfirmFn = Callable[[RoutingDecision], bool]
|
||||
|
||||
|
||||
class RoutingApplicationService:
|
||||
"""Decides which provider/model one turn runs on.
|
||||
|
||||
Args:
|
||||
router: the scoring engine (see :class:`RoutingPort`).
|
||||
mode_reader: ``surface -> mode string``; production passes the per-workspace
|
||||
lookup ``AppContext.project_routing_mode``. Injected rather than read
|
||||
from config here so this layer stays free of config plumbing that
|
||||
EPIC R02 is rewriting in parallel.
|
||||
"""
|
||||
|
||||
def __init__(self, router: RoutingPort,
|
||||
mode_reader: Optional[Callable[[str], str]] = None) -> None:
|
||||
self._router = router
|
||||
self._mode_reader = mode_reader
|
||||
|
||||
# -- main entry point -------------------------------------------------- #
|
||||
def route_turn(
|
||||
self,
|
||||
surface: str,
|
||||
prompt: str,
|
||||
current_provider: str,
|
||||
current_model: str,
|
||||
*,
|
||||
mode: Optional[str] = None,
|
||||
confirm: Optional[ConfirmFn] = None,
|
||||
required_capabilities: Optional[Sequence[str]] = None,
|
||||
task_type: Optional[Any] = None,
|
||||
) -> RoutingDecision:
|
||||
"""Decide what to run this turn on. Never raises.
|
||||
|
||||
A routing failure must never block a message: any unexpected error
|
||||
degrades to "keep the current model", which is exactly what all three
|
||||
legacy copies did with a bare ``except`` - made explicit and testable here.
|
||||
"""
|
||||
resolved_mode = RoutingMode.parse(mode if mode is not None else self._read_mode(surface))
|
||||
keep = self._keep(resolved_mode, current_provider, current_model,
|
||||
reason="routing off - keeping current model")
|
||||
|
||||
# An empty prompt carries no signal to classify, so routing cannot make a
|
||||
# meaningful choice; the same guard exists in all three legacy copies.
|
||||
if resolved_mode is RoutingMode.OFF or not (prompt or "").strip():
|
||||
return keep
|
||||
|
||||
# FALLBACK never switches up front - it only reacts to a failure, which
|
||||
# the caller reports through fallback_after_failure().
|
||||
if resolved_mode is RoutingMode.FALLBACK:
|
||||
return self._keep(resolved_mode, current_provider, current_model,
|
||||
reason="fallback mode - switching only after a failure")
|
||||
|
||||
try:
|
||||
result = self._router.route(
|
||||
surface, prompt, current_provider, current_model,
|
||||
mode_override=resolved_mode.value,
|
||||
required_capabilities=list(required_capabilities) if required_capabilities else None,
|
||||
task_type=task_type,
|
||||
)
|
||||
except Exception: # noqa: BLE001 - routing must never break a turn
|
||||
return self._keep(resolved_mode, current_provider, current_model,
|
||||
reason="routing engine failed - keeping current model")
|
||||
|
||||
proposal = self._to_decision(result, resolved_mode, current_provider, current_model)
|
||||
if not proposal.switched:
|
||||
return proposal
|
||||
|
||||
# Manual mode: the proposal only becomes a switch once a human approves.
|
||||
if resolved_mode is RoutingMode.MANUAL:
|
||||
if confirm is None or not self._ask(confirm, proposal):
|
||||
return self._keep(resolved_mode, current_provider, current_model,
|
||||
reason="switch declined - keeping current model",
|
||||
task_type=proposal.task_type, declined=True)
|
||||
return proposal
|
||||
|
||||
# -- failure recovery -------------------------------------------------- #
|
||||
def fallback_after_failure(
|
||||
self,
|
||||
surface: str,
|
||||
prompt: str,
|
||||
failed_provider: str,
|
||||
failed_model: str,
|
||||
*,
|
||||
mode: Optional[str] = None,
|
||||
required_capabilities: Optional[Sequence[str]] = None,
|
||||
task_type: Optional[Any] = None,
|
||||
) -> Optional[RoutingDecision]:
|
||||
"""Pick a replacement after ``failed_provider/failed_model`` failed.
|
||||
|
||||
Returns None when there is nothing to fall back to, so the caller can
|
||||
surface the original error instead of retrying forever. Available in
|
||||
AUTO and FALLBACK; OFF and MANUAL keep the user's model on failure too,
|
||||
because silently moving work to another model is exactly what those two
|
||||
modes exist to prevent.
|
||||
"""
|
||||
resolved_mode = RoutingMode.parse(mode if mode is not None else self._read_mode(surface))
|
||||
if resolved_mode not in (RoutingMode.AUTO, RoutingMode.FALLBACK):
|
||||
return None
|
||||
|
||||
try:
|
||||
# Asked in AUTO so the engine ranks candidates rather than short-
|
||||
# circuiting on FALLBACK's "never switch up front" rule; the failed
|
||||
# model is passed as current so any positive gain beats it.
|
||||
result = self._router.route(
|
||||
surface, prompt, failed_provider, failed_model,
|
||||
mode_override=RoutingMode.AUTO.value,
|
||||
required_capabilities=list(required_capabilities) if required_capabilities else None,
|
||||
task_type=task_type,
|
||||
)
|
||||
except Exception: # noqa: BLE001 - a broken router must not mask the real error
|
||||
return None
|
||||
|
||||
decision = self._to_decision(result, resolved_mode, failed_provider, failed_model)
|
||||
# A "switch" back to the model that just failed would retry the outage.
|
||||
if not decision.switched or (decision.provider, decision.model) == (failed_provider, failed_model):
|
||||
return None
|
||||
return RoutingDecision(
|
||||
mode=resolved_mode, provider=decision.provider, model=decision.model,
|
||||
switched=True, task_type=decision.task_type, score_gain=decision.score_gain,
|
||||
reason=f"{failed_provider}/{failed_model} failed - falling back to "
|
||||
f"{decision.provider}/{decision.model}",
|
||||
previous_provider=failed_provider, previous_model=failed_model,
|
||||
)
|
||||
|
||||
# -- internals --------------------------------------------------------- #
|
||||
def _read_mode(self, surface: str) -> str:
|
||||
"""Per-surface mode from the injected reader ('off' when none supplied)."""
|
||||
if self._mode_reader is None:
|
||||
return RoutingMode.OFF.value
|
||||
try:
|
||||
return self._mode_reader(surface) or RoutingMode.OFF.value
|
||||
except Exception: # noqa: BLE001 - a config read must not break a turn
|
||||
return RoutingMode.OFF.value
|
||||
|
||||
@staticmethod
|
||||
def _keep(mode: RoutingMode, provider: str, model: str, *, reason: str,
|
||||
task_type: str = "", declined: bool = False) -> RoutingDecision:
|
||||
"""A no-switch decision that still names the model to run on."""
|
||||
return RoutingDecision(mode=mode, provider=provider, model=model, switched=False,
|
||||
task_type=task_type, reason=reason, declined=declined,
|
||||
previous_provider=provider, previous_model=model)
|
||||
|
||||
@staticmethod
|
||||
def _ask(confirm: ConfirmFn, proposal: RoutingDecision) -> bool:
|
||||
"""Run the confirm callback, treating any failure as "declined".
|
||||
|
||||
The callback opens a modal dialog in production; if that raises (window
|
||||
already closing, for instance) the safe answer is to keep the user's own
|
||||
model rather than to switch without consent.
|
||||
"""
|
||||
try:
|
||||
return bool(confirm(proposal))
|
||||
except Exception: # noqa: BLE001
|
||||
return False
|
||||
|
||||
@staticmethod
|
||||
def _to_decision(result: Any, mode: RoutingMode,
|
||||
current_provider: str, current_model: str) -> RoutingDecision:
|
||||
"""Translate the engine's route result into a :class:`RoutingDecision`.
|
||||
|
||||
Defensive about the result shape on purpose: this is the seam between the
|
||||
new layer and a legacy module still under refactor, and a missing
|
||||
attribute must degrade to "keep current model" instead of raising into
|
||||
the middle of a chat turn.
|
||||
"""
|
||||
inner = getattr(result, "decision", None)
|
||||
task_type = getattr(getattr(result, "task_type", None), "value", "") or ""
|
||||
gain = float(getattr(inner, "score_gain", 0.0) or 0.0)
|
||||
reason = str(getattr(inner, "reason", "") or "")
|
||||
|
||||
target = None
|
||||
if getattr(result, "should_switch", False):
|
||||
getter = getattr(result, "target", None)
|
||||
target = getter() if callable(getter) else None
|
||||
|
||||
if not target:
|
||||
return RoutingDecision(mode=mode, provider=current_provider, model=current_model,
|
||||
switched=False, task_type=task_type, score_gain=gain,
|
||||
reason=reason or "no better model - keeping current",
|
||||
previous_provider=current_provider,
|
||||
previous_model=current_model)
|
||||
|
||||
provider, model = target
|
||||
return RoutingDecision(mode=mode, provider=provider or current_provider, model=model,
|
||||
switched=True, task_type=task_type, score_gain=gain, reason=reason,
|
||||
previous_provider=current_provider, previous_model=current_model)
|
||||
|
||||
|
||||
__all__ = ["RoutingApplicationService", "RoutingDecision", "RoutingMode",
|
||||
"RoutingPort", "normalize_mode", "is_valid_mode"]
|
||||
@@ -105,7 +105,7 @@ DEFAULT_CONFIG: Dict[str, Any] = {
|
||||
# sandboxes agent-run shell commands) — reading a URL for info is safe and
|
||||
# useful, so this defaults ON. Toggle in Settings → Security.
|
||||
"allow_url_fetch": True,
|
||||
"sandbox_pw": "quandh14", # default password to unlock sandbox settings
|
||||
"sandbox_pw": "", # set through COWORK_SANDBOX_PASSWORD
|
||||
"rulebase_path": "", # custom RULEBASE.md — attached to every agent execution
|
||||
},
|
||||
# Legacy generic-MCP-server list. MERGED into ext_connectors["other"] as of
|
||||
@@ -173,7 +173,7 @@ DEFAULT_CONFIG: Dict[str, Any] = {
|
||||
# Microsoft. Real Outlook/Teams/OneDrive/SharePoint access still requires a
|
||||
# proper OAuth sign-in (not implemented yet) using tenant_id/client_id below.
|
||||
"ms365": {
|
||||
"unlock_code": "quandh14",
|
||||
"unlock_code": "", # set through COWORK_MS365_UNLOCK_CODE
|
||||
"unlocked": False, # runtime-only — never persisted as True, see save()
|
||||
# Auto-connect MS365/OneDrive/SharePoint: the built-in MS365 MCP server
|
||||
# launches automatically once the user is signed in (OAuth tenant/client
|
||||
@@ -294,6 +294,10 @@ def _apply_env_overrides(data: Dict[str, Any]) -> Dict[str, Any]:
|
||||
data["active_provider"] = os.environ["COWORK_ACTIVE_PROVIDER"]
|
||||
if os.getenv("COWORK_CA_BUNDLE"):
|
||||
data["tls_ca_bundle"] = os.environ["COWORK_CA_BUNDLE"]
|
||||
if os.getenv("COWORK_SANDBOX_PASSWORD"):
|
||||
data["agent_security"]["sandbox_pw"] = os.environ["COWORK_SANDBOX_PASSWORD"]
|
||||
if os.getenv("COWORK_MS365_UNLOCK_CODE"):
|
||||
data["ms365"]["unlock_code"] = os.environ["COWORK_MS365_UNLOCK_CODE"]
|
||||
return data
|
||||
|
||||
|
||||
@@ -552,25 +556,19 @@ class AppConfig:
|
||||
return d
|
||||
|
||||
def routing_mode_for(self, surface: str) -> str:
|
||||
"""Effective Off/Auto/Manual/Fallback mode for a chat surface.
|
||||
|
||||
A per-surface override wins; an empty override falls back to the global
|
||||
``switch_mode``. The value is validated through
|
||||
``application.model_routing.normalize_mode`` so the accepted vocabulary
|
||||
is defined in exactly one place (R03-T03) - it used to be a literal
|
||||
tuple repeated here and in state.py, and adding a mode to one copy but
|
||||
not the others silently downgraded the user's choice to "off"."""
|
||||
from .application.model_routing import normalize_mode
|
||||
"""Effective Off/Auto/Manual mode for a chat surface.
|
||||
|
||||
A per-surface override ("auto"/"manual"/"off") wins; an empty override
|
||||
falls back to the global ``switch_mode``."""
|
||||
routing = self.routing
|
||||
override = (routing.get("surface_modes", {}) or {}).get(surface, "")
|
||||
return normalize_mode(override or routing.get("switch_mode", "off"))
|
||||
mode = override or routing.get("switch_mode", "off")
|
||||
return mode if mode in ("off", "auto", "manual") else "off"
|
||||
|
||||
def set_routing_mode_for(self, surface: str, mode: str) -> None:
|
||||
"""Persist a chat surface's routing toggle selection."""
|
||||
from .application.model_routing import normalize_mode
|
||||
|
||||
self.routing.setdefault("surface_modes", {})[surface] = normalize_mode(mode)
|
||||
"""Persist a chat surface's Off/Auto/Manual toggle selection."""
|
||||
mode = mode if mode in ("off", "auto", "manual") else "off"
|
||||
self.routing.setdefault("surface_modes", {})[surface] = mode
|
||||
self.save()
|
||||
|
||||
@property
|
||||
|
||||
+6
-38
@@ -248,34 +248,9 @@ def _run_agent(ctx, task_type: str, prompt: str, out_dir: Path,
|
||||
"'error' (not silently skip it) if it genuinely can't be completed.\n\n"
|
||||
f"{prompt}"
|
||||
)
|
||||
# One immutable snapshot of this run, then the shared turn service (R04-T05).
|
||||
# The Schedule Task path used to assemble the run_cowork call itself, in
|
||||
# parallel with ui/cowork_tab.py doing the same thing slightly differently -
|
||||
# so a fix to one path silently missed the other. Both now go through
|
||||
# ConversationApplicationService.
|
||||
from ..application.conversations import ConversationApplicationService
|
||||
from ..domain.agents import ConversationExecutionRequest
|
||||
|
||||
messages = [{"role": "user", "content": prompt}]
|
||||
session_id = new_session_id()
|
||||
project_id = project.project_id if project is not None else ""
|
||||
project_context = projects.project_context_text(project)
|
||||
conversation_service = ConversationApplicationService(
|
||||
# The provider was already resolved above (admin agent / per-task
|
||||
# override / machine default), so the factory just hands it back.
|
||||
lambda _provider_id, _model: provider,
|
||||
security_config=ctx.config,
|
||||
)
|
||||
turn = conversation_service.begin_turn(ConversationExecutionRequest.create(
|
||||
prompt, [{"role": "user", "content": prompt}],
|
||||
output_dir=str(out_dir), session_id=session_id, surface="task",
|
||||
title=title, project_id=project_id, project_context=project_context,
|
||||
# Tags every tool call in the audit log as a scheduled task rather than
|
||||
# as the interactive Cowork tab.
|
||||
agent_role=agent_roles.TASK,
|
||||
))
|
||||
# The LIVE list the engine appends to - History is re-saved from it after
|
||||
# every assistant message so a long run shows progress when reopened.
|
||||
messages = turn.messages
|
||||
_save_history_session(ctx, task_type, title, messages, session_id, project_id)
|
||||
# Tell the scheduler the session now genuinely EXISTS on disk — it
|
||||
# refreshes History on this, not on the earlier "task_started" signal
|
||||
@@ -294,21 +269,14 @@ def _run_agent(ctx, task_type: str, prompt: str, out_dir: Path,
|
||||
elif ev.get("type") == "plan_set":
|
||||
last_plan_steps[:] = ev.get("steps") or []
|
||||
|
||||
project_context = projects.project_context_text(project)
|
||||
watched_cancel, timed_out = _cancel_with_timeout(cancel, timeout_sec)
|
||||
try:
|
||||
if task_type == "cowork":
|
||||
# Typed events are rendered back into the legacy dict shape this
|
||||
# module's autosave/plan tracking already consumes; it moves to
|
||||
# AgentEvent directly once the scheduler UI migrates (EPIC R07/R08).
|
||||
result = conversation_service.execute_turn(
|
||||
turn,
|
||||
on_event=lambda event: emit_and_autosave(event.to_dict()),
|
||||
cancel=watched_cancel,
|
||||
)
|
||||
# This module's callers handle a failed run through an exception
|
||||
# (execute_task writes error.txt from it), so re-raise the ORIGINAL
|
||||
# error rather than reporting a silently empty answer.
|
||||
result.raise_if_failed()
|
||||
from .chat_agent import run_cowork
|
||||
run_cowork(provider, messages, out_dir, emit_and_autosave, watched_cancel,
|
||||
security_config=ctx.config, agent_role=agent_roles.TASK,
|
||||
project_context=project_context)
|
||||
else:
|
||||
from .code_agent import run_code
|
||||
limits, block_network = agent_security.sandbox_settings(ctx.config)
|
||||
|
||||
@@ -1,156 +0,0 @@
|
||||
# ADR-001: Kiến Trúc 4 Tầng (Layered / Clean Architecture)
|
||||
|
||||
* **Status**: Accepted
|
||||
* **Date**: 2026-08-21
|
||||
* **EPIC / Task**: R01-T01
|
||||
* **Owner**: 🔵 Team Duy (Tech Lead)
|
||||
* **Áp dụng cho**: toàn bộ mã nguồn mới của `cowork_local` (3 team)
|
||||
|
||||
---
|
||||
|
||||
## 1. Context (Bối cảnh)
|
||||
|
||||
`cowork_local` hiện là một ứng dụng PySide6 desktop local-first ~55.000 dòng Python,
|
||||
được phát triển nhanh theo hướng feature-first. Hệ quả đo được tại thời điểm viết ADR:
|
||||
|
||||
| Vấn đề | Bằng chứng cụ thể trong repo |
|
||||
| :--- | :--- |
|
||||
| **God widget** | `ui/co4e_tab.py` 2.089 dòng, `ui/chat_panel.py` 1.795 dòng, `ui/folder_tab.py` 1.590 dòng |
|
||||
| **Business logic nằm trong widget** | Vòng đời turn chat, quyết định routing, ghép prompt đều nằm trong `ui/chat_panel.py` |
|
||||
| **Logic trùng lặp 3 nơi** | `ui/chat_panel.py::_apply_routing`, `ui/co4e_tab.py::_apply_co4e_routing`, `ui/folder_tab.py::_ai_apply_routing` là ba bản sao gần như y hệt của cùng một thuật toán |
|
||||
| **Không test được nếu không có Qt** | Muốn test một quyết định routing phải dựng widget → không chạy được headless, không chạy được nhanh |
|
||||
| **Side-effect ẩn trong tầng hạ tầng** | Provider tự gọi `core.usage_tracker.record()` ngay trong vòng lặp stream (`providers/openai_compat.py::_record_usage`) |
|
||||
|
||||
Ba team (Duy / Nam / Hoa) sẽ sửa song song trên cùng codebase trong 10 ngày. Nếu
|
||||
không có một ranh giới phụ thuộc được **kiểm chứng tự động**, các thay đổi song song
|
||||
sẽ hội tụ về đúng cấu trúc rối như cũ.
|
||||
|
||||
## 2. Decision (Quyết định)
|
||||
|
||||
Mã nguồn mới được tổ chức thành **4 tầng**, với **chiều phụ thuộc một chiều** như sau:
|
||||
|
||||
```text
|
||||
┌─────────────────────────────────────────────────────────────┐
|
||||
│ presentation/ PySide6 widgets, Qt signals/slots │
|
||||
│ (chat, co4e, workspace…) Chỉ dựng UI và phát/nhận signal │
|
||||
└───────────────────────────┬─────────────────────────────────┘
|
||||
│ gọi xuống (được phép)
|
||||
┌───────────────────────────▼─────────────────────────────────┐
|
||||
│ application/ Pure Python orchestration │
|
||||
│ (conversations, Điều phối use-case, không biết Qt │
|
||||
│ model_routing…) và không biết HTTP/đĩa cụ thể │
|
||||
└───────────────────────────┬─────────────────────────────────┘
|
||||
│ gọi xuống (được phép)
|
||||
┌───────────────────────────▼─────────────────────────────────┐
|
||||
│ domain/ Pure Python entities & events │
|
||||
│ (agents, models…) Frozen dataclass, enum, quy tắc │
|
||||
│ nghiệp vụ thuần. KHÔNG import gì │
|
||||
│ từ 3 tầng còn lại. │
|
||||
└───────────────────────────▲─────────────────────────────────┘
|
||||
│ implement interface của domain
|
||||
┌───────────────────────────┴─────────────────────────────────┐
|
||||
│ infrastructure/ Adapters: network, keyring, đĩa, │
|
||||
│ (providers, telemetry…) process, Qt-free I/O │
|
||||
└─────────────────────────────────────────────────────────────┘
|
||||
```
|
||||
|
||||
### 2.1 Quy tắc bất biến (Invariants)
|
||||
|
||||
| # | Quy tắc | Được kiểm bởi |
|
||||
| :--- | :--- | :--- |
|
||||
| **I1** | `domain/` và `application/` là **100% pure Python** — cấm import `PySide6`, `PyQt5`, `PyQt6`, `shiboken6` | `scripts/check_imports.py` (R01-T03) |
|
||||
| **I2** | `domain/` **không import** `application/`, `infrastructure/`, `presentation/`, `ui/` | `scripts/check_imports.py` |
|
||||
| **I3** | `application/` **không import** `presentation/` hay `ui/` | `scripts/check_imports.py` |
|
||||
| **I4** | Không file production nào vượt **400 dòng** | `scripts/check_loc.py` (R10-T02) |
|
||||
| **I5** | `presentation/` **không** gọi thẳng provider/HTTP/đĩa — phải đi qua một application service | Code review + I1–I3 |
|
||||
| **I6** | Mọi input của một use-case được đóng gói thành **snapshot bất biến** (`frozen dataclass`) trước khi rời UI thread | Code review + unit test |
|
||||
|
||||
### 2.2 Chiều phụ thuộc được phép
|
||||
|
||||
| Từ tầng | Được import | Bị cấm |
|
||||
| :--- | :--- | :--- |
|
||||
| `presentation/` | `application/`, `domain/`, PySide6 | — (nên tránh gọi thẳng `infrastructure/`) |
|
||||
| `application/` | `domain/`, interface do `domain/` định nghĩa | `presentation/`, `ui/`, PySide6 |
|
||||
| `domain/` | chỉ stdlib | tất cả các tầng khác, PySide6 |
|
||||
| `infrastructure/` | `domain/`, thư viện ngoài (requests, keyring…) | `presentation/`, `ui/`, PySide6 |
|
||||
|
||||
### 2.3 Cách tầng dưới "nói chuyện ngược" lên UI
|
||||
|
||||
`application/` **không được** giữ tham chiếu tới widget. Việc trao đổi ngược chiều
|
||||
đi qua **callback thuần Python nhận một `AgentEvent` có kiểu**
|
||||
(`domain/agents/agent_event.py`, R04-T02):
|
||||
|
||||
```python
|
||||
# application layer — pure Python, không biết Qt tồn tại
|
||||
service.run_turn(request, on_event=my_callback)
|
||||
|
||||
# presentation layer — chuyển event sang Qt signal ở ranh giới duy nhất này
|
||||
def my_callback(event: AgentEvent) -> None:
|
||||
self.agent_event.emit(event) # Qt signal → cập nhật UI trên main thread
|
||||
```
|
||||
|
||||
Đây là **seam** duy nhất giữa hai thế giới: dưới seam là Python thuần test được
|
||||
offline, trên seam là Qt. Mọi cập nhật UI phải xảy ra qua Qt signal/slot, không
|
||||
bao giờ gọi trực tiếp từ worker thread.
|
||||
|
||||
## 3. Vị trí sở hữu theo team
|
||||
|
||||
| Tầng / thư mục | Team | EPIC |
|
||||
| :--- | :--- | :--- |
|
||||
| `presentation/chat/`, `application/conversations/`, `application/model_routing/`, `domain/agents/`, `domain/models/`, `infrastructure/providers/`, `infrastructure/telemetry/`, `tests/`, `scripts/` | 🔵 Duy | R01, R03, R04, R08, R10 |
|
||||
| `presentation/co4e/`, `monitoring/`, `settings/`, `shell/`, `application/workflows/`, `infrastructure/config/`, `secrets/`, `sandbox/` | 🟣 Nam | R02, R08, R09 |
|
||||
| `presentation/workspace/`, `folder/`, `scheduling/`, `application/workspaces/`, `scheduling/`, `domain/tools/`, `domain/tasks/`, `infrastructure/filesystem/`, `mcp/`, `persistence/` | 🟢 Hoa | R05, R06, R07, R08 |
|
||||
|
||||
## 4. Chiến lược di trú (Strangler Fig, không big-bang)
|
||||
|
||||
Code cũ trong `core/`, `ui/`, `providers/` **không bị xoá ngay**. Ta bọc dần:
|
||||
|
||||
1. **Tạo seam mới** ở tầng đúng (ví dụ `RoutingApplicationService`).
|
||||
2. **Chuyển call site** cũ sang gọi seam mới (`ui/*.py` chỉ còn vài dòng adapter).
|
||||
3. **Giữ module cũ làm implementation detail** phía sau seam (ví dụ
|
||||
`application/model_routing/` vẫn gọi xuống `core/routing/` để dùng lại
|
||||
scorer/selector đã có test).
|
||||
4. Chỉ khi mọi call site đã đi qua seam mới → cân nhắc gỡ code cũ.
|
||||
|
||||
Nhờ vậy `pytest` luôn xanh giữa các bước, và một team có thể merge mà không chờ
|
||||
team khác refactor xong.
|
||||
|
||||
## 5. Consequences (Hệ quả)
|
||||
|
||||
### Tích cực
|
||||
|
||||
* Test một quyết định routing / một vòng đời turn chat **không cần Qt, không cần mạng** → suite unit chạy < 1 giây.
|
||||
* Ba bản sao logic routing hội tụ về một nơi duy nhất → sửa một lần, cả 3 màn hình cùng đúng.
|
||||
* Người mới có thể thêm một provider mà chỉ chạm `infrastructure/providers/` + `domain/models/`.
|
||||
* Vi phạm kiến trúc bị chặn ở CI thay vì phát hiện lúc review.
|
||||
|
||||
### Tiêu cực / chi phí phải chấp nhận
|
||||
|
||||
* Nhiều file nhỏ hơn thay vì vài file lớn → tăng số lần "nhảy file" khi đọc code.
|
||||
* Tồn tại **hai đường** trong giai đoạn di trú (code cũ + seam mới) cho tới khi call site cuối cùng chuyển xong.
|
||||
* Phải viết DTO/snapshot rõ ràng thay vì truyền thẳng `self` của widget — tốn thêm code, đổi lại được thread-safety.
|
||||
|
||||
## 6. Alternatives considered (Phương án đã cân nhắc)
|
||||
|
||||
| Phương án | Lý do loại |
|
||||
| :--- | :--- |
|
||||
| **Giữ nguyên, chỉ tách file cho ngắn** | Giải quyết được I4 (LOC) nhưng không giải quyết được nguyên nhân gốc: logic vẫn dính Qt nên vẫn không test được offline. |
|
||||
| **MVVM/MVP thuần Qt** | Vẫn buộc business logic phụ thuộc vòng đời Qt object; không chạy được trong scheduler headless và trong task nền. |
|
||||
| **Hexagonal đầy đủ (port/adapter cho mọi thứ)** | Đúng về lý thuyết nhưng quá tốn cho 10 ngày và cho một app desktop 1 process; 4 tầng là điểm cân bằng. |
|
||||
| **Big-bang rewrite** | Rủi ro hồi quy quá cao khi 3 team sửa song song và không có bộ test bảo vệ đầy đủ. |
|
||||
|
||||
## 7. Enforcement (Thực thi)
|
||||
|
||||
```bash
|
||||
python scripts/check_imports.py # I1, I2, I3 — quét AST
|
||||
python scripts/check_loc.py # I4 — giới hạn 400 dòng
|
||||
python scripts/run_quality_gate.py # chạy toàn bộ CASAN Gate + pytest
|
||||
```
|
||||
|
||||
CASAN Verification Gate phải PASS trước khi merge bất kỳ PR nào vào `main`.
|
||||
|
||||
## 8. Tài liệu liên quan
|
||||
|
||||
* `docs/refactor/Feature_Architecture_Proposal.md` — thiết kế tổng thể 10 EPIC
|
||||
* `docs/refactor/Refactoring_Checklist.md` — bảng tiến độ theo task
|
||||
* `docs/architecture/dormant-code.md` — danh mục code không còn hoạt động (R01-T05)
|
||||
@@ -1,85 +0,0 @@
|
||||
# Dormant / Dead Code Inventory (R01-T05)
|
||||
|
||||
* **Task**: R01-T05 — Phân loại và cô lập mã nguồn cũ
|
||||
* **Owner**: 🔵 Team Duy
|
||||
* **Ngày quét**: 2026-08-21
|
||||
* **Phạm vi quét**: toàn bộ `*.py` production (loại trừ `tests/`, `assets/`, `docs/`, `.git/`)
|
||||
|
||||
---
|
||||
|
||||
## 1. Mục đích
|
||||
|
||||
Trước khi 3 team refactor song song, cần biết **file nào thật sự đang chạy**. Refactor
|
||||
một module đã chết là lãng phí; xoá nhầm một module chỉ được gọi động là gây sự cố
|
||||
runtime. Tài liệu này phân loại từng ứng viên, kèm **bằng chứng** và **hành động đề xuất**.
|
||||
|
||||
## 2. Phương pháp
|
||||
|
||||
Quét AST toàn repo, dựng đồ thị import, tìm module **không có module nào khác import**.
|
||||
Kết quả thô: **43 module**. Sau đó xác minh thủ công từng ứng viên, vì phân tích tĩnh
|
||||
không thấy 3 kiểu tham chiếu:
|
||||
|
||||
| Kiểu tham chiếu ẩn | Ví dụ thật trong repo |
|
||||
| :--- | :--- |
|
||||
| Chạy như subprocess | `state.py:285` gọi `python -m cowork_local.mcp_servers.ms365_server` |
|
||||
| Entry point của gói | `__main__.py` (chạy bằng `python -m cowork_local`) |
|
||||
| Script chạy tay | `tools/check_*.py`, `scripts/*.py` |
|
||||
|
||||
> ⚠️ **Kết luận quan trọng**: 43 module "không ai import" **KHÔNG** đồng nghĩa 43 module chết.
|
||||
> Sau xác minh, chỉ còn **6 hạng mục (~1.887 dòng)** là dormant thật.
|
||||
|
||||
## 3. Phân loại kết quả
|
||||
|
||||
### 🟥 A. DORMANT THẬT — không có đường nào chạy tới (ứng viên xoá)
|
||||
|
||||
| Module | LOC | Bằng chứng | Rủi ro khi xoá | Hành động |
|
||||
| :--- | ---: | :--- | :--- | :--- |
|
||||
| `ui/accounts_tab.py` | 700 | Chỉ xuất hiện trong comment của `i18n.py:92`; không widget nào khởi tạo `AccountsTab` | Thấp — panel Monitoring → Accounts hiện không có đường vào | Cô lập, chờ xác nhận PO rồi xoá |
|
||||
| `ui/flow_dialog.py` | 596 | Chỉ được nhắc trong docstring `ui/agent_manager_tab.py:4` và comment `i18n.py:2124` | Trung bình — Flow Manager có thể là tính năng tạm ẩn | **Hỏi PO trước**, chưa xoá |
|
||||
| `security/` (cả package) | 296 | `prompt_validator`, `action_validator`, `attachment_validator`, `audit_logger`, `command_risk_classifier` — không file nào ngoài package tự import. Chức năng **trùng** `core/agent_security.py` + `core/security_rules.py` (đang chạy thật) | Trung bình — dễ nhầm đây là lớp bảo mật đang hoạt động | ⚠️ Ưu tiên cao: xoá hoặc hợp nhất trong **R09 (Team Nam)** |
|
||||
| `core/codebase_memory_ui.py` | 123 | Không nơi nào import; `core/codebase_memory.py` (bản không-UI) mới là bản đang dùng | Thấp | Xoá |
|
||||
| `core/graph_server.py` | 115 | Docstring nói phục vụ build không có QtWebEngine, nhưng **không có call site nào**; `ui/structure_graph_view.py` không gọi | Trung bình — có thể là fallback cho bản .exe chưa nối dây | Xác minh với bản đóng gói PyInstaller trước khi xoá |
|
||||
| `ui/mcp_servers_dialog.py` | 57 | Không import; MCP settings hiện nằm trong `ui/settings_dialog.py` | Thấp | Xoá |
|
||||
|
||||
**Tổng: ~1.887 dòng (≈ 3,4% codebase).**
|
||||
|
||||
### 🟨 B. KHÔNG CHẾT — chạy qua đường ẩn (giữ nguyên)
|
||||
|
||||
| Module | Vì sao phân tích tĩnh báo nhầm |
|
||||
| :--- | :--- |
|
||||
| `__main__.py` | Entry point `python -m cowork_local` |
|
||||
| `mcp_servers/ms365_server.py` | Chạy như tiến trình con — `state.py:285` |
|
||||
| `core/routing/__init__.py` | Được import qua đường dẫn con (`from .routing.service import RoutingService`), heuristic theo tên lá không thấy |
|
||||
| `tools/check_*.py` (34 file, 6.608 dòng) | Bộ smoke-test UI chạy tay: `python tools/check_nav.py`. Là **dev tooling**, không phải code chết |
|
||||
| `scripts/bootstrap_gitea_repo.py`, `scripts/check_imports.py` | Script CLI chạy tay / chạy trong CI |
|
||||
|
||||
### 🟩 C. CODE SỐNG NHƯNG "ĐÓNG BĂNG" — đụng vào phải cẩn thận
|
||||
|
||||
| Module | LOC | Ghi chú cho người refactor |
|
||||
| :--- | ---: | :--- |
|
||||
| `core/chat_agent.py::run_cowork` | 580 | Đang có **characterization test** (`tests/characterization/test_run_cowork.py`, R01-T04). Mọi thay đổi hành vi phải làm cùng lúc với cập nhật snapshot |
|
||||
| `providers/base.py` | 401 | Là contract chung của mọi provider; đổi chữ ký = vỡ cả 3 team. Đã có contract test (R03-T01) |
|
||||
| `core/routing/*` | 2.263 | Đã có 79 test đang xanh. R03 **bọc** chứ không viết lại: `application/model_routing/` gọi xuống đây |
|
||||
|
||||
## 4. Quy tắc xử lý (bắt buộc)
|
||||
|
||||
1. **Không xoá trong cùng PR với refactor.** Xoá code chết là một commit riêng, để `git revert` được độc lập khi có sự cố.
|
||||
2. **Cô lập trước, xoá sau.** Đánh dấu module bằng docstring cảnh báo, chạy 1 vòng release; không ai báo lỗi mới xoá.
|
||||
3. **Hạng mục 🟥 A cần một người xác nhận** (PO hoặc chủ tính năng) trước khi xoá — trừ khi rõ ràng là bản trùng lặp (`codebase_memory_ui`, `mcp_servers_dialog`).
|
||||
4. **Không refactor code trong nhóm 🟥 A.** Nếu một file trong danh sách này >400 dòng, nó **không** tính vào CASAN Check 2 — vì đường đi đúng là xoá, không phải tách nhỏ.
|
||||
|
||||
## 5. Việc cần bàn giao
|
||||
|
||||
| Hạng mục | Team nhận | EPIC |
|
||||
| :--- | :--- | :--- |
|
||||
| `security/` trùng lặp với `core/agent_security.py` | 🟣 Nam | R09 |
|
||||
| `ui/accounts_tab.py`, `ui/flow_dialog.py`, `ui/mcp_servers_dialog.py` | 🟣 Nam (sở hữu `presentation/shell/`, `settings/`) | R08 |
|
||||
| `core/graph_server.py`, `core/codebase_memory_ui.py` | 🟢 Hoa (sở hữu `presentation/graph/`) | R06 |
|
||||
|
||||
## 6. Cách chạy lại lần quét này
|
||||
|
||||
```bash
|
||||
python scripts/check_imports.py # ranh giới kiến trúc (R01-T03)
|
||||
# Bản quét đồ thị import dùng cho tài liệu này sẽ được đóng gói thành
|
||||
# scripts/find_dormant.py trong R10-T02 (Testing & Governance tooling).
|
||||
```
|
||||
@@ -0,0 +1,159 @@
|
||||
# Mô hình chính sách an toàn — CoworkLocal
|
||||
|
||||
R09-T01 · Team Gamma · viết 22/08/2026
|
||||
|
||||
Tài liệu này mô tả **hệ thống đang chạy**, không phải hệ thống mong muốn. Mọi
|
||||
khẳng định đều chỉ tới file và dòng cụ thể để đối chiếu được.
|
||||
|
||||
---
|
||||
|
||||
## 1. Câu hỏi quan trọng nhất: đây có phải rào chắn an ninh không
|
||||
|
||||
**Không.** `core/agent_security.py` nói thẳng ngay ở đầu file:
|
||||
|
||||
> *"this is a business productivity tool, not a hard security boundary"*
|
||||
|
||||
Điều đó quyết định mọi thứ còn lại. Cụ thể: **mọi tầng dùng AI đều mở khi
|
||||
hỏng** (`allowed=True` khi không gọi được validator, `core/agent_security.py:150`).
|
||||
Mạng chập chờn hay gateway trục trặc thì agent vẫn chạy, không bị khoá cứng.
|
||||
|
||||
Đánh đổi có chủ đích: chọn *dùng được* thay vì *chặn tuyệt đối*. Ai đọc tài
|
||||
liệu này để đánh giá rủi ro cần hiểu đúng điều đó — đây là lớp giảm tai nạn,
|
||||
không phải lớp chống kẻ tấn công có chủ đích.
|
||||
|
||||
---
|
||||
|
||||
## 2. Hai loại quy tắc, đừng lẫn
|
||||
|
||||
| | Quy tắc xác định | Quy tắc do AI phán |
|
||||
|---|---|---|
|
||||
| Cách hoạt động | So khớp mẫu cố định | Hỏi một model |
|
||||
| Kết quả | Luôn giống nhau | Có thể khác nhau giữa hai lần |
|
||||
| Khi hỏng | Vẫn chạy | **Mở** (cho qua) |
|
||||
| Tắt được không | Không — luôn bật | Có, từng tầng một |
|
||||
| Ở đâu | Bộ phân loại mẫu chặn + sandbox | 3 tầng validate |
|
||||
|
||||
Câu ở `core/agent_security.py:250` nói rõ ranh giới:
|
||||
|
||||
> *"always-on block-pattern classifier + sandbox still apply regardless"*
|
||||
|
||||
Nghĩa là **tắt hết ba tầng AI thì vẫn còn hai lớp xác định**. Đây là điểm dễ
|
||||
hiểu nhầm nhất khi đọc màn Cài đặt: mấy công tắc ở đó **chỉ tắt phần AI**.
|
||||
|
||||
---
|
||||
|
||||
## 3. Ba tầng AI
|
||||
|
||||
Bật/tắt độc lập trong `agent_security` của `config.json`.
|
||||
|
||||
| Tầng | Kiểm cái gì | Khoá cấu hình | Khi nào chạy |
|
||||
|---|---|---|---|
|
||||
| Prompt | Yêu cầu của chính người dùng | `validate_prompt` | Trước khi agent làm gì |
|
||||
| Attachment | Văn bản trích ra từ tệp đính kèm | `validate_attachments` | Trước khi vào ngữ cảnh model |
|
||||
| Command | `run_command` / `install_package` | `validate_commands` | Trước khi thực thi |
|
||||
|
||||
Cả ba đọc chung một bộ luật: file cục bộ `core/security_rules.py` cộng thêm
|
||||
tài liệu quản trị viên đặt trên OneDrive (nếu có cấu hình). Riêng agent Code
|
||||
dùng bộ luật khác — `RULEforCode.md` thay vì `RULEBASE.md`.
|
||||
|
||||
Công tắc tổng `agent_security.enabled` tắt cả ba.
|
||||
|
||||
---
|
||||
|
||||
## 4. Chuyện gì xảy ra khi bị chặn
|
||||
|
||||
Theo đúng thứ tự trong `core/agent_security.py:266-273`:
|
||||
|
||||
1. Hiện thông báo trong khung chat — người dùng thấy ngay, kèm lý do
|
||||
2. Ghi `audit_log.record("security_block", …)` — vào nhật ký kiểm toán
|
||||
3. `notify_admin(...)` — gửi email quản trị viên
|
||||
4. Ném `SecurityBlocked` — dừng lượt chạy
|
||||
|
||||
Ba bước đầu **không được phép ném lỗi**. `audit_log.record()` có ghi rõ trong
|
||||
docstring: *"never raises — audit logging must never break a chat turn"*. Ghi
|
||||
nhật ký hỏng không được kéo theo cả phiên làm việc.
|
||||
|
||||
---
|
||||
|
||||
## 5. Hỏi người dùng: trạng thái thứ ba
|
||||
|
||||
Ngoài cho/chặn còn một trạng thái nữa mà hệ thống hiện tại **có nhưng chưa gọi
|
||||
tên**: hỏi người dùng.
|
||||
|
||||
`ui/chat_panel.py:1312` kiểm `ctx.project_confirm_commands()` rồi bật
|
||||
`PermissionDialog`. Đó là một quyết định chính sách thật, nhưng nằm rải ở tầng
|
||||
giao diện chứ không phải một kết quả chính thức.
|
||||
|
||||
`domain/security/tool_policy.py` (đề xuất, chờ Team Hoa xác nhận) gộp lại
|
||||
thành ba trạng thái:
|
||||
|
||||
| | Nghĩa |
|
||||
|---|---|
|
||||
| `ALLOW` | Chạy |
|
||||
| `DENY` | Không chạy, có lý do |
|
||||
| `ASK` | Hỏi người dùng đã |
|
||||
|
||||
**`ASK` không phải là `allowed`.** Coi ASK như ALLOW nghĩa là tool chạy trước
|
||||
khi có ai đồng ý — bẫy dễ mắc nhất, đã có test riêng chặn.
|
||||
|
||||
Cổng chính sách **không tự bật hộp thoại**. Nó chỉ trả lời; hỏi ai và hỏi thế
|
||||
nào là việc của tầng giao diện. Nhờ vậy Co4E chạy nền mới dùng chung cổng được
|
||||
với Cowork chạy tương tác — Co4E không hỏi được thì đổi `ASK` thành `DENY`.
|
||||
|
||||
---
|
||||
|
||||
## 6. Bí mật
|
||||
|
||||
Từ 21/08 (R02-T05), API key **không còn nằm trong `config.json`**:
|
||||
|
||||
* Lưu trong kho của hệ điều hành qua `KeyringAdapter` — Windows Credential
|
||||
Manager, macOS Keychain, Linux Secret Service
|
||||
* `provider_conf()` đọc từ kho rồi ghép vào dict trả về, nên chỗ gọi không
|
||||
đổi (đường A, `GammaTeam_decisions.md`)
|
||||
* File cũ tự chuyển ở lần mở đầu tiên, có sao lưu trước khi chuyển
|
||||
|
||||
Máy không có kho bí mật (Linux headless, CI) thì **không chuyển** — thà để
|
||||
khoá trong file còn hơn xoá đi rồi người dùng mất khoá.
|
||||
|
||||
Kiểm bằng `python scripts/audit_security.py`, chạy tự động trong CI.
|
||||
|
||||
---
|
||||
|
||||
## 7. Sandbox
|
||||
|
||||
`core/sandbox_manager.py` chạy lệnh trong môi trường hạn chế. Luôn bật, không
|
||||
tắt được, không phụ thuộc công tắc AI nào.
|
||||
|
||||
Năng lực khác nhau theo hệ điều hành — ma trận đầy đủ sẽ nằm ở
|
||||
`infrastructure/sandbox/sandbox_capabilities.py` (R09-T06, Hiệp phụ trách).
|
||||
Chỗ này cập nhật khi task đó xong.
|
||||
|
||||
---
|
||||
|
||||
## 8. Những chỗ đã biết là yếu
|
||||
|
||||
Ghi ra để người sau khỏi tưởng đã kín:
|
||||
|
||||
1. **Mở khi hỏng.** Gateway chết là ba tầng AI cho qua hết. Có chủ đích, nhưng
|
||||
nghĩa là không chống được kẻ tấn công biết cách làm validator ngừng trả lời.
|
||||
2. **Bí mật vẫn đi trong bộ nhớ.** Đường A ghép khoá vào dict `provider_conf()`
|
||||
trả về, nên khoá vẫn có thể lọt vào log gỡ lỗi hay ảnh chụp màn hình. Đường
|
||||
B (bỏ hẳn khỏi dict) đã ghi vào nợ kỹ thuật.
|
||||
3. **Bộ luật lấy từ OneDrive không ký số.** Ai sửa được tài liệu đó là sửa được
|
||||
luật.
|
||||
4. **`ASK` chưa được nối vào Co4E.** Co4E chạy nền, chưa có đường hỏi người
|
||||
dùng — hiện phải chọn giữa cho qua hết hoặc chặn hết.
|
||||
|
||||
---
|
||||
|
||||
## Đối chiếu nhanh
|
||||
|
||||
| Nội dung | Nguồn |
|
||||
|---|---|
|
||||
| Ba tầng AI, mở khi hỏng | `core/agent_security.py:1-25` |
|
||||
| Phân loại mẫu + sandbox luôn bật | `core/agent_security.py:250` |
|
||||
| Thứ tự khi bị chặn | `core/agent_security.py:266-273` |
|
||||
| Nhật ký không được ném lỗi | `core/audit_log.py:46` |
|
||||
| Hỏi người dùng | `ui/chat_panel.py:1312` |
|
||||
| Ba trạng thái chính sách | `domain/security/tool_policy.py` |
|
||||
| Bí mật | `infrastructure/secrets/keyring_adapter.py` |
|
||||
@@ -0,0 +1,58 @@
|
||||
# Project Context MCP — hướng dẫn làm song song
|
||||
|
||||
Mục tiêu: hoàn thiện ba tool trên **cùng một server** `project_context`. Không tạo server, registry,
|
||||
policy hay error envelope mới. Shared skeleton đã khóa sẵn thứ tự an toàn:
|
||||
|
||||
```text
|
||||
validate input → policy ALLOW → resolve provider → gọi upstream → validate output
|
||||
```
|
||||
|
||||
## Chia việc
|
||||
|
||||
| Người | Tool | Chỉ sửa | Branch đề xuất |
|
||||
|---|---|---|---|
|
||||
| Member A | `get_project_issue_context` | `tools/issue_context.py`, `providers/issue.py`, test riêng | `feat/mcp-issue-context` |
|
||||
| Member B | `search_project_knowledge` | `tools/knowledge_search.py`, `providers/knowledge.py`, test riêng | `feat/mcp-knowledge-search` |
|
||||
| Member C | `get_project_change_context` | `tools/change_context.py`, `providers/change.py`, test riêng | `feat/mcp-change-context` |
|
||||
|
||||
Trước khi gửi task, thay `Member A/B/C` bằng username thật trên ba issue. Mỗi người **không sửa**
|
||||
`foundation.py`, `registry.py`, `runtime.py`, `server.py` hoặc file của người khác. Nếu shared contract
|
||||
cần đổi, mở một PR nhỏ riêng và để cả ba người rebase sau khi PR đó merge.
|
||||
|
||||
## Bắt đầu trong 5 phút
|
||||
|
||||
1. Chạy `python --version` và xác nhận Python 3.11+ như baseline trong `requirements.txt`.
|
||||
2. Tạo branch từ commit template chứa tài liệu này sau khi PR template merge.
|
||||
3. Đọc input/output model trong module tool được giao; không thêm field riêng của Gitea/Jira/Redmine.
|
||||
4. Implement provider read-only trong module `providers/<tool>.py`; credential chỉ lấy sau policy ALLOW.
|
||||
5. Thêm test happy, invalid, not-found, timeout, DENIED với `resolver.calls == 0`, output sai schema,
|
||||
truncation/cursor và source mở được có `revision`.
|
||||
6. Chạy:
|
||||
|
||||
```bash
|
||||
python -m pytest tests/test_project_context_mcp_template.py tests/test_project_context_<tool>.py -q
|
||||
```
|
||||
|
||||
Lệnh trên chạy trực tiếp từ root repo `cowork_local`; `tests/conftest.py` đã thiết lập import path.
|
||||
|
||||
## Definition of Done của từng người
|
||||
|
||||
- Tool trả đúng schema, có `project_id` và source gồm `system`, `url`, `revision`, `retrieved_at`.
|
||||
- Provider-neutral: đổi Gitea sang GitHub/Jira/Redmine không đổi schema hay tool name.
|
||||
- Sai project bị `DENIED` trước khi resolve credential và trước mọi upstream call.
|
||||
- Không log/return token; lỗi ngoài dự kiến không lộ exception; read không có side effect.
|
||||
- Output lớn có `truncated`, `returned`, `remaining`, `next_cursor`; không cắt im lặng.
|
||||
- Test riêng pass, test shared pass, PR chỉ chạm đúng vùng sở hữu trong bảng trên.
|
||||
|
||||
## Chạy server sau khi provider đã cấu hình
|
||||
|
||||
```bash
|
||||
COWORK_MCP_ACTOR_ID=<actor> \
|
||||
COWORK_MCP_ORG_UNIT=<org> \
|
||||
COWORK_MCP_CUSTOMER=<customer> \
|
||||
COWORK_MCP_PROJECT=<project> \
|
||||
python -m cowork_local.mcp_servers.project_context_server
|
||||
```
|
||||
|
||||
Không commit giá trị môi trường hoặc credential. Cowork kết nối bằng stdio với command Python và
|
||||
args `-m cowork_local.mcp_servers.project_context_server`.
|
||||
@@ -1,233 +0,0 @@
|
||||
# BÁO CÁO KẾT QUẢ — TEAM DUY: EPIC R01, R03, R04
|
||||
|
||||
* **Dự án**: Cowork Local (Cowork-Local BamBOO)
|
||||
* **Team**: 🔵 Team Duy — Core AI, Routing, Turn Runtime & Testing (Tech Lead)
|
||||
* **Nhánh**: `feature/deltateam/refactor-plan`
|
||||
* **Thời gian thực hiện**: 21/08/2026, 09:56 ➔ 10:56
|
||||
* **Ngày báo cáo**: 21/08/2026
|
||||
* **Tài liệu gốc**: `Feature_Architecture_Proposal.md`, `Refactoring_Checklist.md`, `DeltaTeam_prompt.md`
|
||||
|
||||
---
|
||||
|
||||
## 1. Tóm tắt điều hành
|
||||
|
||||
Hoàn tất **16/16 task** của 3 EPIC được giao trong đợt này: **R01** (nền tảng kiến trúc & lưới an toàn), **R03** (hợp nhất provider & routing), **R04** (vòng đời turn hội thoại). Toàn bộ đã commit và push lên nhánh.
|
||||
|
||||
| Chỉ số | Kết quả |
|
||||
| :--- | :--- |
|
||||
| Task hoàn thành | **16/16** (R01: 5, R03: 6, R04: 5) |
|
||||
| Commit | 5 |
|
||||
| File thay đổi | 48 (37 file mới, 11 file sửa) |
|
||||
| Dòng code | +5.843 / −225 |
|
||||
| Test | **243 pass** / 44s |
|
||||
| Test suite nhanh (unit + contract + characterization + routing) | **218 pass / 1,22s** |
|
||||
| CASAN Check 3 (`scripts/check_imports.py`) | **PASS** — 0 Qt import trong `domain/`, `application/` |
|
||||
| File production > 400 dòng | **0** |
|
||||
|
||||
**3 lỗi thật được phát hiện và sửa trong quá trình làm** (chi tiết mục 5) — trong đó 1 lỗi deadlock sẽ làm treo ứng dụng ngay ở tin nhắn đầu tiên.
|
||||
|
||||
---
|
||||
|
||||
## 2. Kết quả theo từng EPIC
|
||||
|
||||
### 🔹 EPIC R01 — Architecture Foundation & Characterization (5/5)
|
||||
|
||||
| Task | Sản phẩm | Ghi chú |
|
||||
| :--- | :--- | :--- |
|
||||
| R01-T01 | `docs/architecture/ADR-001-layered-architecture.md` | Định nghĩa 4 tầng, chiều phụ thuộc, 6 quy tắc bất biến I1–I6, chiến lược di trú Strangler Fig |
|
||||
| R01-T02 | `tests/fakes/fake_provider.py`, `fake_tool_executor.py` | Test double chạy offline, kịch bản hoá, ghi lại mọi lời gọi |
|
||||
| R01-T03 | `scripts/check_imports.py` (239 dòng) | Quét AST, bắt cả import tương đối (`from ...ui import x`) và import trong thân hàm |
|
||||
| R01-T04 | `tests/characterization/test_run_cowork.py` | **13 test** chụp snapshot hành vi hiện tại của `run_cowork` trước khi R04 đụng vào |
|
||||
| R01-T05 | `docs/architecture/dormant-code.md` | Quét đồ thị import: 43 module "không ai import" ➔ xác minh còn **6 hạng mục chết thật (~1.887 dòng)** |
|
||||
|
||||
**Điểm đáng chú ý ở R01-T03**: dùng AST thay vì `grep` là bắt buộc — trong repo có nhiều docstring nhắc tên `PySide6` một cách hợp lệ, `grep` sẽ báo nhầm và đội sẽ học cách tắt cổng kiểm duyệt.
|
||||
|
||||
**Điểm đáng chú ý ở R01-T05**: 43 module không có importer **không** đồng nghĩa 43 module chết. Sau xác minh thủ công: `__main__.py` là entry point, `mcp_servers/ms365_server.py` chạy bằng subprocess (`state.py:285`), 34 file `tools/check_*.py` là dev tooling chạy tay. Chỉ 6 hạng mục là dormant thật.
|
||||
|
||||
### 🔹 EPIC R03 — Model Providers & Routing (6/6)
|
||||
|
||||
| Task | Sản phẩm | Ghi chú |
|
||||
| :--- | :--- | :--- |
|
||||
| R03-T01 | `tests/contracts/test_providers.py` | **29 contract test**; chạy được cả 2 adapter thật mà **không cần mạng** nhờ thay `Provider._request` bằng SSE đóng hộp |
|
||||
| R03-T02 | `domain/models/provider_descriptor.py`, `infrastructure/providers/provider_registry.py` | Gom 3 nơi khai báo provider về 1 chỗ |
|
||||
| R03-T03 | `application/model_routing/routing_application_service.py` | Pure Python, 4 chế độ: Off / Auto / Manual / **Fallback (mới)** |
|
||||
| R03-T04, T05 | `ui/chat_panel.py`, `ui/co4e_tab.py`, `ui/folder_tab.py` | Gỡ 3 bản sao logic routing |
|
||||
| R03-T06 | `infrastructure/telemetry/usage_sink.py` | Tách ghi nhận token usage khỏi provider |
|
||||
|
||||
**Vấn đề gốc đã giải quyết** — cùng một thuật toán routing tồn tại **3 bản gần giống nhau**:
|
||||
|
||||
```
|
||||
ui/chat_panel.py::_apply_routing (~45 dòng)
|
||||
ui/co4e_tab.py::_apply_co4e_routing (~38 dòng)
|
||||
ui/folder_tab.py::_ai_apply_routing (~42 dòng)
|
||||
```
|
||||
|
||||
Cả 3 đều nằm trong widget Qt ➔ **không thể test nếu không dựng cửa sổ**, và đã bắt đầu lệch nhau (mỗi bản xác định "model hiện tại" một kiểu). Nay cả 3 chỉ còn gọi `ctx.routing_application().route_turn(...)` + một callback xác nhận.
|
||||
|
||||
**Chế độ Fallback (mới)**: giữ nguyên model người dùng chọn, **chỉ đổi sau khi model đó lỗi**. Đây là chế độ người dùng cần khi họ tin lựa chọn của mình nhưng vẫn muốn lượt chat sống sót qua sự cố nhà cung cấp.
|
||||
|
||||
**Bộ từ vựng mode**: trước đây tuple `("off", "auto", "manual")` bị lặp ở **4 chỗ** (`config.py` × 2, `state.py` × 2). Thêm một mode mà quên một chỗ sẽ **âm thầm hạ lựa chọn của người dùng về "off"**. Nay tập trung vào `normalize_mode()` / `is_valid_mode()`.
|
||||
|
||||
### 🔹 EPIC R04 — Agent Runtime & Conversation Service (5/5)
|
||||
|
||||
| Task | Sản phẩm | Ghi chú |
|
||||
| :--- | :--- | :--- |
|
||||
| R04-T01 | `domain/agents/conversation_execution_request.py` | Frozen dataclass, chụp toàn bộ input của 1 turn tại thời điểm submit |
|
||||
| R04-T02 | `domain/agents/agent_event.py` (370 dòng) | **13 event có kiểu** thay cho dict không kiểu, kèm cầu nối 2 chiều |
|
||||
| R04-T03 | `application/conversations/conversation_application_service.py` | Điều phối vòng đời turn, không import Qt |
|
||||
| R04-T04 | `ui/cowork_tab.py::build_job` | Chuyển sang snapshot + service |
|
||||
| R04-T05 | `core/task_executors.py::_run_agent` | Chuyển sang **cùng** service (trước đây là bản lắp ráp thứ hai, hơi khác) |
|
||||
|
||||
**Vấn đề gốc đã giải quyết** — closure trong `build_job` đọc state của widget **từ trong worker thread**:
|
||||
|
||||
```python
|
||||
def job(worker):
|
||||
provider = self.build_provider() # đọc combo box
|
||||
proj_ctx = project_context_text(load_project(project_id))
|
||||
```
|
||||
|
||||
Người dùng có thể đổi model, đổi workspace, sửa chỉ dẫn project **trong lúc turn đang chạy**. Turn khi đó chạy trên hỗn hợp state cũ + mới, và hỗn hợp nào phụ thuộc vào thời điểm luồng — đúng loại bug tái hiện mỗi tuần một lần và không bao giờ tái hiện trong test.
|
||||
|
||||
**`TurnCompletedEvent`** là tín hiệu kết thúc turn mà engine cũ **hoàn toàn không có**: hiện tại mọi consumer suy ra "xong" từ việc worker thread kết thúc, nên **turn bị huỷ và turn thất bại trông giống hệt nhau** với giao diện.
|
||||
|
||||
---
|
||||
|
||||
## 3. Kiến trúc sau refactor
|
||||
|
||||
```text
|
||||
presentation/ ui/chat_panel.py, ui/co4e_tab.py, ui/folder_tab.py, ui/cowork_tab.py
|
||||
│ (chỉ dựng UI, mở dialog xác nhận, render thông báo)
|
||||
▼
|
||||
application/ model_routing/routing_application_service.py ← 4 mode routing
|
||||
conversations/conversation_application_service.py ← vòng đời turn
|
||||
│ (100% pure Python — cổng kiểm duyệt tự động chặn import Qt)
|
||||
▼
|
||||
domain/ agents/conversation_execution_request.py ← snapshot bất biến
|
||||
agents/agent_event.py ← 13 event có kiểu
|
||||
models/provider_descriptor.py ← catalog provider
|
||||
▲
|
||||
infrastructure/ providers/provider_registry.py telemetry/usage_sink.py
|
||||
```
|
||||
|
||||
**Nguyên tắc di trú (ADR-001 mục 4)**: **không viết lại engine**. `core/chat_agent.py::run_cowork` và `core/routing/*` (2.263 dòng, 79 test đang xanh) vẫn là engine bên dưới; tầng application chỉ sở hữu phần trước đây bị trộn vào UI. Nhờ vậy `pytest` luôn xanh giữa các bước và một team có thể merge mà không phải chờ team khác.
|
||||
|
||||
---
|
||||
|
||||
## 4. Bằng chứng kiểm thử
|
||||
|
||||
### Phân bố test
|
||||
|
||||
| Suite | Số test | Thời gian | Vai trò |
|
||||
| :--- | ---: | ---: | :--- |
|
||||
| `tests/unit/` | 97 | | Logic thuần, không Qt/mạng |
|
||||
| `tests/contracts/` | 29 | | Mọi provider phải thoả cùng bộ cam kết |
|
||||
| `tests/characterization/` | 13 | | Chốt hành vi hiện tại của `run_cowork` |
|
||||
| `tests/routing/` | 79 | | Có sẵn từ trước, vẫn xanh |
|
||||
| **Cộng 4 suite nhanh** | **218** | **1,22s** | ✅ đạt CASAN "A — unit < 1s" |
|
||||
| `tests/integration/` | 25 | 42s | Widget Qt thật (offscreen) + provider kịch bản hoá |
|
||||
| **Tổng** | **243** | **44s** | |
|
||||
|
||||
### Đối chiếu Definition of Done (7 tiêu chí, `DeltaTeam_prompt.md`)
|
||||
|
||||
| # | Tiêu chí | Kết quả |
|
||||
| :--- | :--- | :--- |
|
||||
| 1 | Mọi file < 400 dòng | ✅ Lớn nhất: `agent_event.py` 370 dòng |
|
||||
| 2 | 0 import Qt trong `domain/`, `application/` | ✅ `check_imports.py` PASS |
|
||||
| 3 | Comment tiếng Anh ở mọi khối sửa/mới | ✅ Docstring + giải thích **lý do**, không chỉ mô tả code |
|
||||
| 4 | Có unit/contract test, pass 100% < 1s | ✅ 218 test / 1,22s |
|
||||
| 5 | Không hồi quy | ✅ 79 test routing có sẵn vẫn xanh |
|
||||
| 6 | Ghi Start/End vào Checklist | ✅ 16 task đã tick kèm mốc thời gian |
|
||||
| 7 | Cổng CASAN | ⚠️ `run_quality_gate.py` thuộc **R10-T02**, chưa viết. Check 3 đã có và PASS |
|
||||
|
||||
### Ba đường code đã sửa nhưng ban đầu chưa được thực thi
|
||||
|
||||
Sau khi hoàn tất 16 task, rà soát lại phát hiện 3 đường code đã bị sửa nhưng **không test nào chạy qua**. Đã bổ sung **18 test**:
|
||||
|
||||
| Đường code | Rủi ro nếu bỏ qua | Test bổ sung |
|
||||
| :--- | :--- | ---: |
|
||||
| `task_executors._run_agent` | Autosave History có thể đóng băng ở tin nhắn đầu | 7 |
|
||||
| `_apply_co4e_routing` / `_ai_apply_routing` | Mới chỉ import được, chưa từng gọi hàm | 11 |
|
||||
| `confirm_switch(decision)` Manual mode | Thiếu field ➔ **nổ bên trong modal**, nơi khó phát hiện nhất | (nằm trong 11 ở trên) |
|
||||
|
||||
---
|
||||
|
||||
## 5. Ba lỗi thật phát hiện trong quá trình làm
|
||||
|
||||
### 🔴 Lỗi 1 — Deadlock khi khởi tạo routing service
|
||||
|
||||
`AppContext.routing_application()` giữ `_routing_lock` rồi gọi `routing()`, vốn cũng lấy **chính lock đó**. `threading.Lock` không reentrant ➔ **treo cứng ngay ở tin nhắn đầu tiên**, không có thông báo lỗi.
|
||||
|
||||
*Sửa*: tách `_routing_app_lock` riêng, và resolve engine **trước khi** lấy lock.
|
||||
|
||||
### 🟠 Lỗi 2 — Event `notice` bị cầu nối nuốt mất
|
||||
|
||||
Bản đầu của `agent_event.py` liệt kê 12 loại event nhưng **thiếu `notice`**. Trong khi đó `notice` được phát ra từ 3 nơi trên đường chạy bình thường:
|
||||
|
||||
* `core/agent_security.py` — yêu cầu/lệnh bị Agent Security **chặn**
|
||||
* `core/context_budget.py` — hội thoại vừa bị tự động nén
|
||||
* Bộ đọc file đính kèm — file không xử lý được, và tiến độ "đang đọc trang X/Y"
|
||||
|
||||
Cầu nối bỏ qua event không nhận diện được (đúng thiết kế, để engine có thể thêm event mới) — nên **người dùng sẽ không bao giờ thấy cảnh báo bảo mật**, hoàn toàn im lặng.
|
||||
|
||||
*Sửa*: thêm `NoticeEvent`, **và** thêm test quét mã nguồn engine tìm mọi tag `emit({"type": ...})` rồi bắt lỗi nếu có tag nào chưa có event tương ứng — biến sự im lặng thành test đỏ.
|
||||
|
||||
### 🟡 Lỗi 3 — Test đang chạy trên checkout khác
|
||||
|
||||
`tests/routing/conftest.py` đẩy thư mục cha vào `sys.path`. Vì thư mục checkout tên là `cowork_local_gitea` (không phải `cowork_local`), lệnh `import cowork_local` **ăn nhầm sang `Desktop\cowork_local`** — một bản checkout khác. Suite báo xanh trên mã nguồn **không phải nhánh đang review**.
|
||||
|
||||
*Sửa*: `tests/conftest.py` nạp `__init__.py` theo đường dẫn tuyệt đối và đăng ký vào `sys.modules` trước mọi test.
|
||||
|
||||
---
|
||||
|
||||
## 6. Cải thiện phụ (không nằm trong yêu cầu task)
|
||||
|
||||
| Cải thiện | Ảnh hưởng |
|
||||
| :--- | :--- |
|
||||
| `ProviderRegistry.build()` đóng dấu `descriptor.id` lên instance | Sửa việc usage của `ollama` / `github_copilot` / `codex` bị ghi nhận nhầm thành `openai_compat` trên Dashboard. **Chưa nối vào production** — xem mục 7. |
|
||||
| `ProviderRegistry.build()` copy config trước khi ghi | Trước đây một model do routing chọn có thể ghi đè lên default đã lưu của người dùng |
|
||||
| `UsageTrackerSink` ghi log ở mức debug khi thất bại | Trước là `except: pass` — mất sạch lý do khi Dashboard hỏng |
|
||||
| `estimate_tokens` được chốt bằng test so với `core.usage_tracker` | Bảo đảm việc tách telemetry **không làm lệch một con số nào** |
|
||||
|
||||
---
|
||||
|
||||
## 7. Còn nợ & cần quyết định
|
||||
|
||||
| # | Nội dung | Người quyết |
|
||||
| :--- | :--- | :--- |
|
||||
| 1 | **`ProviderRegistry` chưa nối vào `state.build_provider_for`** (vẫn dùng `providers/factory.py`). Nối vào sẽ sửa lỗi quy kết usage ở mục 6, **nhưng đổi cách gom dữ liệu lịch sử trên Dashboard**. | Team Duy + PO |
|
||||
| 2 | **Mode `fallback` chưa có trên toggle UI** — config và service đã hỗ trợ đầy đủ; widget `RoutingToggle` thuộc R08. | Team Duy (R08) |
|
||||
| 3 | **Đã sửa 2 dòng trong `config.py`** (`routing_mode_for`, `set_routing_mode_for`) để dùng chung bộ từ vựng mode. File này Team Nam đang refactor ở R02-T02. | ⚠️ **Cần báo Team Nam** |
|
||||
| 4 | **Circular import** `core/model_pricing.py` ↔ `core/usage_tracker.py` chưa xử lý (task ngày 28/08). | Team Duy |
|
||||
| 5 | **2 test đỏ có sẵn từ trước**: `config.py:108` hardcode `sandbox_pw = "quandh14"` ➔ `tests/test_config_security.py`. Thuộc **EPIC R02 / Team Nam**. | 🟣 Team Nam |
|
||||
| 6 | `tests/integration/test_routing_surfaces.py` mất 41s do dựng `Co4ETab`/`FolderTab`. Nên gắn marker `slow` khi làm R10. | Team Duy (R10) |
|
||||
|
||||
---
|
||||
|
||||
## 8. Phạm vi chưa kiểm thử
|
||||
|
||||
Nêu rõ để tránh hiểu nhầm mức độ bảo đảm:
|
||||
|
||||
* **Chưa mở ứng dụng bằng tay** — mới chạy widget headless (`QT_QPA_PLATFORM=offscreen`), chưa có ai kiểm tra bằng mắt.
|
||||
* **Chưa gọi provider thật** — toàn bộ dùng `FakeProvider`, không có lưu lượng mạng.
|
||||
* **Chưa chạy 34 script `tools/check_*.py`** — các script này tự `sys.path.insert` thư mục cha nên sẽ import nhầm checkout khác (đúng lỗi 3 ở mục 5). Cần sửa chúng ở R10.
|
||||
|
||||
---
|
||||
|
||||
## 9. Việc kế tiếp của Team Duy
|
||||
|
||||
| EPIC | Nội dung | Điều kiện |
|
||||
| :--- | :--- | :--- |
|
||||
| **R08** (T01 ➔ T06) | Tách `ui/chat_panel.py` (1.795 dòng) thành 6 widget < 400 dòng | Sẵn sàng bắt đầu — `AgentEvent` (R04-T02) chính là kênh dữ liệu 6 widget con sẽ dùng thay vì đọc trực tiếp state của `ChatPanel` |
|
||||
| **R10** (T01 ➔ T05) | Testing Pyramid, `run_quality_gate.py`, Contributor Recipes, E2E Smoke | Chờ cả 3 team hoàn tất |
|
||||
|
||||
---
|
||||
|
||||
## 10. Lịch sử commit
|
||||
|
||||
| Commit | Nội dung |
|
||||
| :--- | :--- |
|
||||
| `bbc09f6` | feat(R01): architecture foundation, offline fakes and characterization net |
|
||||
| `96bec97` | feat(R03): unify provider catalogue, routing decisions and usage telemetry |
|
||||
| `a53163e` | feat(R04): immutable turn snapshot, typed agent events, conversation service |
|
||||
| `15e1d3e` | test(R03/R04): cover the three code paths that were changed but never executed |
|
||||
| `67b8d2e` | docs(refactor): correct the Team Duy scope block in the checklist |
|
||||
@@ -0,0 +1,768 @@
|
||||
<!doctype html>
|
||||
<html lang="vi">
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1">
|
||||
<title>Phân Việc Refactor Team Gamma</title>
|
||||
</head>
|
||||
<body>
|
||||
<style>
|
||||
:root {
|
||||
--ground: #FAF9FC; --surface: #FFFFFF; --surface-2: #F3F0F8;
|
||||
--ink: #191325; --muted: #665E7C; --line: #E4DEEE;
|
||||
--accent: #6D3A9E;
|
||||
--lead: #6D3A9E; --m1: #14707F; --m2: #A65418;
|
||||
--warn: #A81F1A; --ok: #1B6B40;
|
||||
--lead-wash: #F1E9F9; --m1-wash: #E2F1F3; --m2-wash: #F8EDE2;
|
||||
}
|
||||
@media (prefers-color-scheme: dark) {
|
||||
:root:not([data-theme="light"]) {
|
||||
--ground: #121020; --surface: #1B1830; --surface-2: #241F3D;
|
||||
--ink: #EFEBF7; --muted: #A69EBD; --line: #2F2848;
|
||||
--accent: #C08CF0;
|
||||
--lead: #C08CF0; --m1: #56C6D8; --m2: #E29A56;
|
||||
--warn: #F08A84; --ok: #6FD69C;
|
||||
--lead-wash: #2A1F42; --m1-wash: #133038; --m2-wash: #38270F;
|
||||
}
|
||||
}
|
||||
:root[data-theme="dark"] {
|
||||
--ground: #121020; --surface: #1B1830; --surface-2: #241F3D;
|
||||
--ink: #EFEBF7; --muted: #A69EBD; --line: #2F2848;
|
||||
--accent: #C08CF0;
|
||||
--lead: #C08CF0; --m1: #56C6D8; --m2: #E29A56;
|
||||
--warn: #F08A84; --ok: #6FD69C;
|
||||
--lead-wash: #2A1F42; --m1-wash: #133038; --m2-wash: #38270F;
|
||||
}
|
||||
|
||||
* { box-sizing: border-box; }
|
||||
body {
|
||||
margin: 0; background: var(--ground); color: var(--ink);
|
||||
font-family: "Segoe UI", -apple-system, system-ui, "Helvetica Neue", sans-serif;
|
||||
font-size: 15px; line-height: 1.62; -webkit-font-smoothing: antialiased;
|
||||
}
|
||||
.wrap { max-width: 1120px; margin: 0 auto; padding: 0 28px 96px; }
|
||||
code, .mono, td.day, th.day, .num {
|
||||
font-family: Consolas, "Cascadia Mono", "SF Mono", ui-monospace, monospace;
|
||||
font-variant-numeric: tabular-nums;
|
||||
}
|
||||
|
||||
header.top { border-bottom: 2px solid var(--ink); padding: 56px 0 22px; margin-bottom: 34px; }
|
||||
.eyebrow { font-size: 12px; letter-spacing: .16em; text-transform: uppercase;
|
||||
color: var(--accent); font-weight: 700; margin: 0 0 14px; }
|
||||
h1 { font-size: clamp(30px, 4.4vw, 44px); line-height: 1.1; margin: 0 0 16px;
|
||||
font-weight: 700; letter-spacing: -.02em; text-wrap: balance; }
|
||||
.lede { font-size: 17px; color: var(--muted); margin: 0; max-width: 64ch; }
|
||||
.alias { margin: 18px 0 0; padding: 12px 16px; max-width: 74ch;
|
||||
background: var(--surface-2); border-left: 3px solid var(--accent);
|
||||
border-radius: 3px; font-size: 14px; color: var(--muted); }
|
||||
.alias b { color: var(--ink); }
|
||||
.facts { display: flex; flex-wrap: wrap; gap: 28px; margin-top: 26px;
|
||||
padding-top: 20px; border-top: 1px solid var(--line); }
|
||||
.fact .k { font-size: 11px; letter-spacing: .13em; text-transform: uppercase;
|
||||
color: var(--muted); display: block; margin-bottom: 3px; }
|
||||
.fact .v { font-size: 15px; font-weight: 600; }
|
||||
|
||||
h2 { font-size: 23px; margin: 52px 0 6px; letter-spacing: -.01em; font-weight: 700; text-wrap: balance; }
|
||||
h2 + .sub { color: var(--muted); margin: 0 0 22px; max-width: 70ch; }
|
||||
h3 { font-size: 17px; margin: 30px 0 10px; font-weight: 700; }
|
||||
|
||||
/* gate = việc phải xong trước khi chia nhánh */
|
||||
.gatebox { background: var(--surface); border: 1px solid var(--line);
|
||||
border-left: 4px solid var(--warn); border-radius: 3px; padding: 4px 26px 22px; }
|
||||
.gatebox h2 { margin-top: 22px; }
|
||||
|
||||
.steps { list-style: none; counter-reset: s; padding: 0; margin: 0; }
|
||||
.steps > li { counter-increment: s; position: relative; padding: 14px 0 14px 46px;
|
||||
border-bottom: 1px solid var(--line); }
|
||||
.steps > li:last-child { border-bottom: none; }
|
||||
.steps > li::before {
|
||||
content: counter(s); position: absolute; left: 0; top: 14px;
|
||||
width: 26px; height: 26px; border-radius: 50%; background: var(--accent);
|
||||
color: #fff; font-size: 13px; font-weight: 700; display: flex;
|
||||
align-items: center; justify-content: center;
|
||||
font-family: Consolas, ui-monospace, monospace;
|
||||
}
|
||||
.steps b { display: block; margin-bottom: 2px; }
|
||||
.steps small { color: var(--muted); font-size: 13.5px; display: block; }
|
||||
.est { float: right; font-size: 12px; color: var(--muted); font-weight: 600;
|
||||
font-family: Consolas, ui-monospace, monospace; }
|
||||
|
||||
.cards { display: grid; grid-template-columns: repeat(3, 1fr); gap: 18px; }
|
||||
@media (max-width: 940px) { .cards { grid-template-columns: 1fr; } }
|
||||
.card { background: var(--surface); border: 1px solid var(--line);
|
||||
border-top: 3px solid var(--c); border-radius: 3px; padding: 20px;
|
||||
display: flex; flex-direction: column; }
|
||||
.card.lead { --c: var(--lead); --w: var(--lead-wash); }
|
||||
.card.one { --c: var(--m1); --w: var(--m1-wash); }
|
||||
.card.two { --c: var(--m2); --w: var(--m2-wash); }
|
||||
.card .tag { font-size: 11px; letter-spacing: .13em; text-transform: uppercase;
|
||||
font-weight: 700; color: var(--c); margin-bottom: 6px; }
|
||||
.card h3 { margin: 0 0 4px; font-size: 18px; }
|
||||
.card .who { font-size: 13px; color: var(--muted); margin-bottom: 14px; }
|
||||
.card .branch { font-size: 12.5px; background: var(--w); color: var(--c);
|
||||
padding: 5px 9px; border-radius: 3px; display: inline-block;
|
||||
margin-bottom: 16px; word-break: break-all; font-weight: 600; }
|
||||
.card h4 { font-size: 11px; letter-spacing: .12em; text-transform: uppercase;
|
||||
color: var(--muted); margin: 16px 0 7px; font-weight: 700; }
|
||||
.card ul { margin: 0; padding-left: 17px; font-size: 14px; }
|
||||
.card li { margin-bottom: 6px; }
|
||||
.tid { font-size: 12px; font-weight: 700; color: var(--c);
|
||||
font-family: Consolas, ui-monospace, monospace; }
|
||||
.paths { list-style: none; padding: 0; margin: 0; font-size: 12.5px; }
|
||||
.paths li { padding: 3px 0; border-bottom: 1px dotted var(--line);
|
||||
font-family: Consolas, ui-monospace, monospace; color: var(--muted); word-break: break-all; }
|
||||
.paths li:last-child { border-bottom: none; }
|
||||
.weight { margin-top: auto; padding-top: 16px; font-size: 12.5px; color: var(--muted); }
|
||||
.weight b { color: var(--ink); font-size: 15px; }
|
||||
|
||||
.scroll { overflow-x: auto; border: 1px solid var(--line); border-radius: 3px; }
|
||||
table { border-collapse: collapse; width: 100%; font-size: 13.5px; background: var(--surface); }
|
||||
th, td { text-align: left; padding: 11px 14px; border-bottom: 1px solid var(--line); vertical-align: top; }
|
||||
thead th { background: var(--surface-2); font-size: 11px; letter-spacing: .1em;
|
||||
text-transform: uppercase; color: var(--muted); font-weight: 700; white-space: nowrap; }
|
||||
tbody tr:last-child td { border-bottom: none; }
|
||||
td.day, th.day { white-space: nowrap; font-weight: 700; font-size: 13px; }
|
||||
td.cl { border-left: 3px solid var(--lead); }
|
||||
td.c1 { border-left: 3px solid var(--m1); }
|
||||
td.c2 { border-left: 3px solid var(--m2); }
|
||||
tr.mark td { background: var(--surface-2); font-weight: 600; }
|
||||
td small { color: var(--muted); display: block; font-size: 12.5px; }
|
||||
.pill { display: inline-block; font-size: 11px; font-weight: 700; padding: 2px 7px;
|
||||
border-radius: 2px; letter-spacing: .04em; white-space: nowrap; }
|
||||
.pill.cp { background: var(--m1-wash); color: var(--m1); }
|
||||
.pill.gate { background: var(--m2-wash); color: var(--m2); }
|
||||
.pill.ship { background: var(--lead-wash); color: var(--lead); }
|
||||
|
||||
.rules { display: grid; grid-template-columns: repeat(2, 1fr); gap: 16px; }
|
||||
@media (max-width: 760px) { .rules { grid-template-columns: 1fr; } }
|
||||
.rule { background: var(--surface); border: 1px solid var(--line);
|
||||
border-radius: 3px; padding: 18px 20px; border-left: 3px solid var(--c, var(--line)); }
|
||||
.rule.hard { --c: var(--warn); }
|
||||
.rule.soft { --c: var(--ok); }
|
||||
.rule h3 { margin: 0 0 8px; font-size: 15px; }
|
||||
.rule p { margin: 0; font-size: 14px; color: var(--muted); }
|
||||
.rule code { color: var(--ink); }
|
||||
|
||||
|
||||
/* tóm tắt: đọc 30 giây là nắm được, trước khi vào chi tiết */
|
||||
.tldr {
|
||||
display: grid; grid-template-columns: 1.35fr 1fr; gap: 0;
|
||||
border: 1px solid var(--line); border-radius: 3px; overflow: hidden;
|
||||
margin-bottom: 8px; background: var(--surface);
|
||||
}
|
||||
@media (max-width: 820px) { .tldr { grid-template-columns: 1fr; } }
|
||||
.tldr > div { padding: 20px 24px; }
|
||||
.tldr .right { background: var(--surface-2); border-left: 1px solid var(--line); }
|
||||
@media (max-width: 820px) { .tldr .right { border-left: none; border-top: 1px solid var(--line); } }
|
||||
.tldr .cap {
|
||||
font-size: 11px; letter-spacing: .14em; text-transform: uppercase;
|
||||
color: var(--muted); font-weight: 700; margin: 0 0 12px;
|
||||
}
|
||||
.flow { list-style: none; padding: 0; margin: 0; font-size: 14px; }
|
||||
.flow li { padding: 7px 0; border-bottom: 1px dotted var(--line); display: flex; gap: 10px; }
|
||||
.flow li:last-child { border-bottom: none; }
|
||||
.flow .b {
|
||||
flex: 0 0 auto; font-size: 11.5px; font-weight: 700; padding: 1px 7px; border-radius: 2px;
|
||||
background: var(--w2); color: var(--c2); height: fit-content; margin-top: 2px;
|
||||
font-family: Consolas, ui-monospace, monospace;
|
||||
}
|
||||
.flow li.f0 { --c2: var(--warn); --w2: var(--surface-2); }
|
||||
.flow li.f1 { --c2: var(--lead); --w2: var(--lead-wash); }
|
||||
.flow li.f2 { --c2: var(--m1); --w2: var(--m1-wash); }
|
||||
.flow li.f3 { --c2: var(--m2); --w2: var(--m2-wash); }
|
||||
.flow .t { flex: 1; }
|
||||
.flow .t b { display: block; }
|
||||
.flow .t small { color: var(--muted); font-size: 12.5px; }
|
||||
.must { margin: 0; padding-left: 18px; font-size: 14px; }
|
||||
.must li { margin-bottom: 8px; }
|
||||
.must li:last-child { margin-bottom: 0; }
|
||||
.must b { color: var(--ink); }
|
||||
|
||||
/* input / output từng người */
|
||||
.io { display: grid; gap: 18px; }
|
||||
.iorow { background: var(--surface); border: 1px solid var(--line);
|
||||
border-left: 3px solid var(--c); border-radius: 3px; overflow: hidden; }
|
||||
.iorow.n1 { --c: var(--lead); --w: var(--lead-wash); }
|
||||
.iorow.n2 { --c: var(--m1); --w: var(--m1-wash); }
|
||||
.iorow.n3 { --c: var(--m2); --w: var(--m2-wash); }
|
||||
.iohead { padding: 14px 20px; background: var(--w); display: flex;
|
||||
align-items: baseline; gap: 12px; flex-wrap: wrap; }
|
||||
.iohead b { color: var(--c); font-size: 15px; }
|
||||
.iohead span { color: var(--muted); font-size: 13px; }
|
||||
.iogrid { display: grid; grid-template-columns: 1fr 1fr; }
|
||||
@media (max-width: 860px) { .iogrid { grid-template-columns: 1fr; } }
|
||||
.iogrid > div { padding: 16px 20px; }
|
||||
.iogrid > div + div { border-left: 1px solid var(--line); }
|
||||
@media (max-width: 860px) {
|
||||
.iogrid > div + div { border-left: none; border-top: 1px solid var(--line); }
|
||||
}
|
||||
.iocap { font-size: 11px; letter-spacing: .13em; text-transform: uppercase;
|
||||
color: var(--muted); font-weight: 700; margin: 0 0 10px; }
|
||||
.iolist { list-style: none; margin: 0; padding: 0; font-size: 13.5px; }
|
||||
.iolist li { padding: 5px 0; border-bottom: 1px dotted var(--line); }
|
||||
.iolist li:last-child { border-bottom: none; }
|
||||
.iolist code { font-size: 12.5px; }
|
||||
.frm { display: inline-block; font-size: 11px; font-weight: 700; padding: 1px 6px;
|
||||
border-radius: 2px; background: var(--surface-2); color: var(--muted);
|
||||
margin-right: 6px; font-family: Consolas, ui-monospace, monospace; }
|
||||
.frm.done { background: var(--m1-wash); color: var(--m1); }
|
||||
.frm.risk { background: var(--m2-wash); color: var(--m2); }
|
||||
|
||||
footer { margin-top: 64px; padding-top: 20px; border-top: 1px solid var(--line);
|
||||
font-size: 13px; color: var(--muted); }
|
||||
</style>
|
||||
|
||||
<div class="wrap">
|
||||
|
||||
<header class="top">
|
||||
<p class="eyebrow">Team Gamma · Automation, Workflows & Governance</p>
|
||||
<h1>Một nhánh chung, ba làn không đụng nhau</h1>
|
||||
<p class="lede">
|
||||
Toàn bộ phần việc refactor 10 ngày của Team Gamma — Nam, Hiệp, Lâm. Cả ba đẩy chung vào <code>gamma/refactor</code>. Nam làm thêm một mục chung —
|
||||
khung kiến trúc, hợp đồng dữ liệu, cổng kiểm duyệt — nằm ngoài ba nhánh; xong mục đó thì
|
||||
ba người vào ba nhánh tính năng ngang nhau, không ai phải sửa chung file với ai.
|
||||
</p>
|
||||
<p class="alias">
|
||||
Ba tài liệu refactor gọi team này là <b>“Team Nam”</b> (theo tên lead). Cùng một team, cùng
|
||||
phạm vi R02 · R08 · R09 · R07-T06. Nhánh của team dùng tiền tố <code>gamma/</code>; ba tài liệu refactor viết
|
||||
<code>nam/workflow-governance-*</code> theo tên lead — cùng một thứ.
|
||||
</p>
|
||||
<div class="facts">
|
||||
<div class="fact"><span class="k">Thời hạn</span><span class="v mono">21/08 → 31/08</span></div>
|
||||
<div class="fact"><span class="k">Người</span><span class="v mono">Nam · Hiệp · Lâm</span></div>
|
||||
<div class="fact"><span class="k">Nhánh</span><span class="v mono">gamma/refactor</span></div>
|
||||
<div class="fact"><span class="k">Code phải bóc</span><span class="v mono">~6.500 dòng</span></div>
|
||||
<div class="fact"><span class="k">Cổng phải qua</span><span class="v mono">CASAN Check 1</span></div>
|
||||
</div>
|
||||
</header>
|
||||
|
||||
|
||||
<section class="tldr">
|
||||
<div>
|
||||
<p class="cap">Tóm tắt · thứ tự làm</p>
|
||||
<ul class="flow">
|
||||
<li class="f0">
|
||||
<span class="b">CHUNG</span>
|
||||
<span class="t"><b>Nam làm trước, nửa ngày</b>
|
||||
<small>Dựng khung 5 thư mục (đang là 0 file) · interface + fake cho Config/Secrets ·
|
||||
chốt <code>api_key</code> và báo Team Duy · script CASAN Check 1 · đưa 3 check vào CI ·
|
||||
quyết số phận 24 checker UI. Merge xong mới chia nhánh.</small></span>
|
||||
</li>
|
||||
<li class="f1">
|
||||
<span class="b">N1</span>
|
||||
<span class="t"><b>N1 — Nam · Cấu hình, Bí mật, Vỏ ứng dụng</b>
|
||||
<small>R02 (6 task) · settings 4 widget · bootstrap + MainWindow · policy doc.
|
||||
Giữ luôn <code>app.py</code>, <code>config.py</code>, <code>theme.py</code>,
|
||||
<code>i18n.py</code>. ~2.700 dòng.</small></span>
|
||||
</li>
|
||||
<li class="f2">
|
||||
<span class="b">N2</span>
|
||||
<span class="t"><b>N2 — Hiệp · Giám sát</b>
|
||||
<small>7 tab Monitoring · CanonicalAuditLogger · MonitoringQueryService ·
|
||||
2 vòng lặp import · ma trận Sandbox. ~2.650 dòng.</small></span>
|
||||
</li>
|
||||
<li class="f3">
|
||||
<span class="b">N3</span>
|
||||
<span class="t"><b>N3 — Lâm · Co4E Studio</b>
|
||||
<small>Co4EWorkflowService · tách <code>co4e_tab.py</code> + <code>co4e_canvas.py</code>
|
||||
thành 5 phần. ~2.880 dòng, file to nhất team.</small></span>
|
||||
</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div class="right">
|
||||
<p class="cap">Ba điều bắt buộc</p>
|
||||
<ol class="must">
|
||||
<li><b>Không chạm file dùng chung.</b> Cần thêm chuỗi hay màu thì nhắn nhóm trưởng, đừng tự sửa.</li>
|
||||
<li><b>Nộp factory, không tự lắp vào <code>app.py</code>.</b> N1 lắp trong <code>bootstrap.py</code> ngày 28/08.</li>
|
||||
<li><b>Bị chặn thì dùng fake, báo ngay trong ngày.</b> Không ngồi đợi ai.</li>
|
||||
</ol>
|
||||
<p class="cap" style="margin-top:20px">Nghiệm thu</p>
|
||||
<p style="margin:0;font-size:14px;color:var(--muted)">
|
||||
Trên <code>gamma/refactor</code>: không file nào được sửa bởi hai người khác nhau.
|
||||
Có là quy ước <b style="color:var(--ink)">số 1</b> đang bị vi phạm.
|
||||
</p>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section class="gatebox">
|
||||
<h2>Mục chung — Nam làm, xong hai người kia mới bắt đầu</h2>
|
||||
<p class="sub">
|
||||
Sáu việc dưới đây không thuộc làn nào — chúng là thứ cả ba người cùng đụng
|
||||
vào. Nam làm một lần và đẩy lên <code>gamma/refactor</code>, rồi hai người kia
|
||||
mới bắt đầu. Ước tính nửa ngày.
|
||||
</p>
|
||||
<ol class="steps">
|
||||
<li>
|
||||
<span class="est">~30 phút</span>
|
||||
<b>Dựng khung thư mục</b>
|
||||
<small>
|
||||
<code>domain/</code> <code>application/</code> <code>infrastructure/</code>
|
||||
<code>presentation/</code> <code>platform/</code> <code>tests/fakes/</code> —
|
||||
hiện tại <b>chưa tồn tại, 0 file</b>. Mọi task của cả ba người đều ghi vào đây; để ba
|
||||
người tự tạo là đụng nhau ở <code>__init__.py</code> ngay ngày đầu.
|
||||
</small>
|
||||
</li>
|
||||
<li>
|
||||
<span class="est">~45 phút</span>
|
||||
<b>Viết interface + fake cho Config và Secrets</b>
|
||||
<small>
|
||||
<code>SecretStore</code>, <code>ConfigRepository</code>, kèm
|
||||
<code>FakeSecretStore</code> và <code>FakeConfigRepository</code>. Chỉ chữ ký, chưa cần
|
||||
thân hàm. Đây là thứ gỡ chốt cho cả hai người kia — <b>156 lời gọi
|
||||
<code>ctx.config.*</code> trong 29 file</b> đang chờ nó.
|
||||
</small>
|
||||
</li>
|
||||
<li>
|
||||
<span class="est">~20 phút</span>
|
||||
<b>Chốt số phận <code>api_key</code> và báo Team Duy</b>
|
||||
<small>
|
||||
<code>provider_conf()</code> còn trả <code>api_key</code> bên trong, hay tách hẳn sang
|
||||
<code>SecretStore</code>? Có 5 nơi đọc trực tiếp, <b>3 trong số đó nằm trong
|
||||
<code>providers/</code> của Team Duy</b>. Quyết một mình rồi im lặng là làm vỡ code
|
||||
team bạn.
|
||||
</small>
|
||||
</li>
|
||||
<li>
|
||||
<span class="est">~30 phút</span>
|
||||
<b>Viết <code>scripts/audit_security.py</code> (CASAN Check 1)</b>
|
||||
<small>
|
||||
Gamma chủ trì check này ngày 30/08. Viết ngay hôm nay thì lead tự kiểm được trong suốt
|
||||
quá trình chuyển API key, thay vì tới ngày cổng mới chạy lần đầu và phát hiện vấn đề.
|
||||
</small>
|
||||
</li>
|
||||
<li>
|
||||
<span class="est">~20 phút</span>
|
||||
<b>Thêm 3 check CASAN vào CI</b>
|
||||
<small>
|
||||
CI hiện chỉ chạy <code>pytest tests -q</code>. Ba check (secret · ≤400 dòng · import
|
||||
guard) không nằm trong CI, nên tới 30/08 mới biết ai vi phạm. Đưa vào CI thì mỗi PR tự
|
||||
báo.
|
||||
</small>
|
||||
</li>
|
||||
<li>
|
||||
<span class="est">~30 phút</span>
|
||||
<b>Quyết số phận 24 checker UI, rồi thông báo</b>
|
||||
<small>
|
||||
Chúng bám vào <code>cowork_local.config</code> (34 chỗ) và <code>cowork_local.app</code>
|
||||
(16 chỗ) — <b>sẽ chết ngay khi lead đụng <code>config.py</code></b>. Đây là lưới an toàn
|
||||
duy nhất cho phần UI vừa làm xong. Xem mục quy ước bên dưới.
|
||||
</small>
|
||||
</li>
|
||||
</ol>
|
||||
</section>
|
||||
|
||||
<h2>Ba làn</h2>
|
||||
<p class="sub">
|
||||
Ba làn ngang nhau, mỗi làn khoảng 2.700 dòng phải bóc tách, <b>cùng đẩy vào một
|
||||
nhánh</b> <code>gamma/refactor</code>. Nam nhận làn N1 vì đó là làn chạm tới file
|
||||
dùng chung nhiều nhất. Cột “sở hữu” là danh sách file <em>chỉ</em> người đó được
|
||||
sửa — trên nhánh chung, đây là thứ duy nhất giữ cho ba người không giẫm chân.
|
||||
</p>
|
||||
|
||||
<div class="cards">
|
||||
|
||||
<div class="card lead">
|
||||
<div class="tag">Làn N1 · Nam</div>
|
||||
<h3>Cấu hình, Bí mật & Vỏ ứng dụng</h3>
|
||||
<p class="who">Nam giữ — làn chạm nhiều file dùng chung nhất</p>
|
||||
<div class="branch">gamma/refactor</div>
|
||||
|
||||
<h4>Việc</h4>
|
||||
<ul>
|
||||
<li><span class="tid">R02-T01…T06</span> AtomicJsonFile · ConfigRepository · Typed Settings Facade · SecretStore + Keyring · chuyển API key · schema versioning</li>
|
||||
<li><span class="tid">R08-T07</span> tách <code>settings_dialog.py</code> → 4 section widget</li>
|
||||
<li><span class="tid">R08-T10</span> <code>bootstrap.py</code> + tách <code>MainWindow</code> → shell · tray · lifecycle <em>(cuối sprint, lắp factory của hai người kia)</em></li>
|
||||
<li><span class="tid">R09-T01</span> tài liệu Security Policy Model</li>
|
||||
<li>Chủ trì <b>CASAN Check 1</b> · giữ CI · duyệt PR của hai người</li>
|
||||
</ul>
|
||||
|
||||
<h4>Sở hữu độc quyền</h4>
|
||||
<ul class="paths">
|
||||
<li>config.py</li>
|
||||
<li>app.py → presentation/shell/</li>
|
||||
<li>bootstrap.py</li>
|
||||
<li>theme.py · i18n.py</li>
|
||||
<li>infrastructure/config/ · secrets/ · persistence/</li>
|
||||
<li>ui/settings_dialog.py → presentation/settings/</li>
|
||||
<li>scripts/ · .gitea/workflows/</li>
|
||||
</ul>
|
||||
|
||||
<p class="weight"><b>~2.700 dòng</b> · 727 settings + 1.352 app + 616 config<br>+ mục chung ở trên</p>
|
||||
</div>
|
||||
|
||||
<div class="card one">
|
||||
<div class="tag">Làn N2 · Hiệp</div>
|
||||
<h3>Giám sát & Quan trắc</h3>
|
||||
<p class="who">Hiệp — 7 tab, việc lặp cần kỷ luật</p>
|
||||
<div class="branch">gamma/refactor</div>
|
||||
|
||||
<h4>Việc</h4>
|
||||
<ul>
|
||||
<li><span class="tid">R08-T08</span> tách <code>monitoring_tab.py</code> → 7 tab độc lập</li>
|
||||
<li><span class="tid">R09-T04</span> <code>CanonicalAuditLogger</code></li>
|
||||
<li><span class="tid">R09-T05</span> <code>MonitoringQueryService</code> read-only, phân trang</li>
|
||||
<li><span class="tid">R09-T02</span> gỡ vòng lặp <code>model_pricing</code> ↔ <code>usage_tracker</code></li>
|
||||
<li><span class="tid">R09-T03</span> gỡ vòng lặp <code>agent_security</code> ↔ <code>alert</code></li>
|
||||
<li><span class="tid">R09-T06</span> ma trận Sandbox theo hệ điều hành</li>
|
||||
</ul>
|
||||
|
||||
<h4>Sở hữu độc quyền</h4>
|
||||
<ul class="paths">
|
||||
<li>ui/monitoring_tab.py → presentation/monitoring/</li>
|
||||
<li>application/monitoring/</li>
|
||||
<li>infrastructure/telemetry/ · sandbox/</li>
|
||||
<li>core/audit_log.py</li>
|
||||
<li>core/model_pricing.py · usage_tracker.py</li>
|
||||
<li>core/agent_security*.py</li>
|
||||
</ul>
|
||||
|
||||
<p class="weight"><b>~2.650 dòng</b> · 1.545 monitoring + ~1.100 core</p>
|
||||
</div>
|
||||
|
||||
<div class="card two">
|
||||
<div class="tag">Làn N3 · Lâm</div>
|
||||
<h3>Co4E Studio</h3>
|
||||
<p class="who">Lâm — canvas và luồng chạy workflow</p>
|
||||
<div class="branch">gamma/refactor</div>
|
||||
|
||||
<h4>Việc</h4>
|
||||
<ul>
|
||||
<li><span class="tid">R07-T06</span> <code>Co4EWorkflowService</code> thuần Python</li>
|
||||
<li><span class="tid">R08-T09</span> tách <code>co4e_tab.py</code> + <code>co4e_canvas.py</code> → canvas · node property · run control · chat view · agent list</li>
|
||||
<li>Gọi tool qua <code>ToolPolicyGateway</code> của Team Hoa — dùng fake, không chờ</li>
|
||||
</ul>
|
||||
|
||||
<h4>Sở hữu độc quyền</h4>
|
||||
<ul class="paths">
|
||||
<li>ui/co4e_tab.py → presentation/co4e/</li>
|
||||
<li>ui/co4e_canvas.py</li>
|
||||
<li>ui/co4e_config_panel.py</li>
|
||||
<li>application/workflows/</li>
|
||||
<li>domain/workflows/</li>
|
||||
<li>core/co4e_run_manager.py</li>
|
||||
</ul>
|
||||
|
||||
<p class="weight"><b>~2.880 dòng</b> · file to nhất của cả team</p>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
|
||||
<h2>Tám quy ước</h2>
|
||||
<p class="sub">
|
||||
Tám điều dưới đây là luật của team, Nam chốt. Bốn điều đầu là bắt buộc — trên một
|
||||
nhánh chung, vi phạm không chỉ hại mình mà chặn cả hai người kia.
|
||||
</p>
|
||||
|
||||
<div class="rules">
|
||||
|
||||
<div class="rule hard">
|
||||
<h3>1 · Không chạm file dùng chung</h3>
|
||||
<p>
|
||||
<code>app.py</code>, <code>theme.py</code>, <code>i18n.py</code>, <code>config.py</code>,
|
||||
<code>bootstrap.py</code> thuộc nhánh N1 của Nam. Cần thêm chuỗi hay token màu thì
|
||||
<b>nhắn, đừng sửa</b> — Nam thêm trong ngày. Đây là ba file duy nhất có thể gây conflict
|
||||
thật, và luật này xoá hẳn khả năng đó.
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div class="rule hard">
|
||||
<h3>2 · Nộp factory, không tự lắp vào app</h3>
|
||||
<p>
|
||||
Mỗi nhánh expose một hàm dựng widget với chữ ký chốt từ ngày đầu, ví dụ
|
||||
<code>build_monitoring_tab(ctx, query_service) -> QWidget</code>. Nam gọi nó trong <code>bootstrap.py</code> ngày 28/08. Không ai tự sửa chỗ khởi tạo trong
|
||||
<code>app.py</code>.
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div class="rule hard">
|
||||
<h3>3 · Bị chặn thì dùng fake, không ngồi đợi</h3>
|
||||
<p>
|
||||
Chưa có <code>ConfigRepository</code> bản thật thì dùng <code>FakeConfigRepository</code>.
|
||||
Chưa có <code>ToolPolicyGateway</code> của Team Hoa thì đã có fake sẵn. <b>Báo ngay trong ngày</b>
|
||||
nếu thiếu fake nào — đó là việc của nhóm trưởng, không phải lý do dừng tay.
|
||||
</p>
|
||||
</div>
|
||||
|
||||
|
||||
<div class="rule hard">
|
||||
<h3>4 · Nhánh chung: kéo trước khi đẩy, đừng để nhánh đỏ</h3>
|
||||
<p>
|
||||
Cả ba đẩy vào <code>gamma/refactor</code>, nên không còn nhánh riêng làm vùng
|
||||
đệm. Ba việc bắt buộc: <code>git pull --rebase</code> trước mỗi lần đẩy;
|
||||
commit nhỏ và đẩy trong ngày, đừng ôm 500 dòng ba hôm; và
|
||||
<b>không bao giờ đẩy thứ làm <code>pytest tests -q</code> đỏ</b> — nhánh hỏng
|
||||
là hai người kia đứng hình. Lỡ đẩy nhầm thì sửa ngay hoặc
|
||||
<code>git revert</code>, đừng để qua đêm.
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div class="rule soft">
|
||||
<h3>5 · Commit nhỏ, mỗi ngày một lần</h3>
|
||||
<p>
|
||||
Một PR cho một sub-widget hoặc một service, không dồn 7 tab vào một PR cuối tuần. Nhóm
|
||||
trưởng duyệt trong ngày. PR càng to thì rủi ro càng dồn về ngày 28/08.
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div class="rule soft">
|
||||
<h3>6 · Mỗi commit kèm test, và không làm đỏ 90 test cũ</h3>
|
||||
<p>
|
||||
Baseline hiện tại: <b>102 test xanh trong 3,4 giây</b>. Chạy <code>pytest tests -q</code>
|
||||
trước mỗi lần đẩy. Đây là lưới an toàn cho phần logic — giữ nó xanh suốt 10 ngày.
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div class="rule soft">
|
||||
<h3>7 · File mới ≤ 400 dòng, không import PySide6 vào lõi</h3>
|
||||
<p>
|
||||
Hai điều kiện của CASAN Check 2 và 3. Tự kiểm trước khi đẩy — CI sẽ báo, nhưng biết
|
||||
sớm thì đỡ phải tách lại lần hai.
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div class="rule soft">
|
||||
<h3>8 · Checker UI thuộc phạm vi ai, người đó cập nhật</h3>
|
||||
<p>
|
||||
24 checker sẽ vỡ khi file bị dời. Ai dời file thì sửa checker tương ứng ngay trong
|
||||
commit đó — tốn thêm khoảng 15% thời gian, đổi lại giữ được lưới an toàn cho phần
|
||||
UI vừa làm xong.
|
||||
<b>Đã chốt 21/08: đường A. Nam chịu trách nhiệm nếu đổi ý.</b>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
|
||||
|
||||
<h2>Mỗi người nhận gì, giao gì</h2>
|
||||
<p class="sub">
|
||||
Cột trái là thứ phải có trong tay mới làm được, kèm nguồn. Cột phải là thứ bắt
|
||||
buộc giao ra, kèm người nhận. Nhãn <span class="frm done">có rồi</span> nghĩa là
|
||||
mục chung đã làm xong.
|
||||
</p>
|
||||
|
||||
<div class="io">
|
||||
|
||||
<div class="iorow n1">
|
||||
<div class="iohead"><b>N1 · Cấu hình, Bí mật & Vỏ</b><span>Nam · nhóm trưởng</span></div>
|
||||
<div class="iogrid">
|
||||
<div>
|
||||
<p class="iocap">Input — cần có</p>
|
||||
<ul class="iolist">
|
||||
<li><span class="frm">mã cũ</span><code>config.py</code> 616 dòng</li>
|
||||
<li><span class="frm">mã cũ</span><code>ui/settings_dialog.py</code> 727 dòng</li>
|
||||
<li><span class="frm">mã cũ</span><code>app.py</code> 1.352 dòng</li>
|
||||
<li><span class="frm risk">tự chốt</span>Quyết định <code>api_key</code> — trước 26/08</li>
|
||||
<li><span class="frm">từ Hiệp</span>Chữ ký <code>build_monitoring_tab()</code> — trước 28/08</li>
|
||||
<li><span class="frm">từ Lâm</span>Chữ ký <code>build_co4e_tab()</code> — trước 28/08</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div>
|
||||
<p class="iocap">Output — phải giao</p>
|
||||
<ul class="iolist">
|
||||
<li><span class="frm done">có rồi</span><code>SecretStore</code> · <code>ConfigRepository</code> + fake → <b>cho Hiệp và Lâm</b></li>
|
||||
<li><span class="frm done">có rồi</span><code>scripts/audit_security.py</code> → cho CI</li>
|
||||
<li><code>infrastructure/persistence/json/atomic_json_file.py</code></li>
|
||||
<li><code>infrastructure/config/</code> — cài đặt thật + settings facade</li>
|
||||
<li><code>infrastructure/secrets/keyring_adapter.py</code></li>
|
||||
<li><code>presentation/settings/</code> — 4 widget</li>
|
||||
<li><code>bootstrap.py</code> + <code>presentation/shell/</code> — 3 file</li>
|
||||
<li><code>docs/architecture/security-policy.md</code></li>
|
||||
</ul>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="iorow n2">
|
||||
<div class="iohead"><b>N2 · Giám sát</b><span>Hiệp</span></div>
|
||||
<div class="iogrid">
|
||||
<div>
|
||||
<p class="iocap">Input — cần có</p>
|
||||
<ul class="iolist">
|
||||
<li><span class="frm">mã cũ</span><code>ui/monitoring_tab.py</code> 1.545 dòng</li>
|
||||
<li><span class="frm">mã cũ</span><code>core/usage_tracker.py</code> 524 · <code>sandbox_manager.py</code> 335</li>
|
||||
<li><span class="frm">mã cũ</span><code>core/model_pricing.py</code> 284 · <code>agent_security.py</code> 272 · <code>audit_log.py</code> 115</li>
|
||||
<li><span class="frm done">từ Nam</span><code>FakeConfigRepository</code> — dùng được ngay</li>
|
||||
<li><span class="frm risk">tự chốt</span>Giữ nguyên 9 trường log, báo Duy và Hoa</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div>
|
||||
<p class="iocap">Output — phải giao</p>
|
||||
<ul class="iolist">
|
||||
<li><code>build_monitoring_tab()</code> → <b>cho Nam</b>, trước 28/08</li>
|
||||
<li><code>FakeAuditLogger</code> · <code>FakeMonitoringQueryService</code> → <b>cho cả team</b></li>
|
||||
<li><code>presentation/monitoring/</code> — 7 tab + shell</li>
|
||||
<li><code>application/monitoring/monitoring_query_service.py</code></li>
|
||||
<li><code>infrastructure/telemetry/audit_logger.py</code></li>
|
||||
<li><code>infrastructure/sandbox/sandbox_capabilities.py</code></li>
|
||||
<li><b>0 circular import</b> ở pricing ↔ usage và security ↔ alert</li>
|
||||
</ul>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="iorow n3">
|
||||
<div class="iohead"><b>N3 · Co4E Studio</b><span>Lâm</span></div>
|
||||
<div class="iogrid">
|
||||
<div>
|
||||
<p class="iocap">Input — cần có</p>
|
||||
<ul class="iolist">
|
||||
<li><span class="frm">mã cũ</span><code>ui/co4e_tab.py</code> 2.089 dòng</li>
|
||||
<li><span class="frm">mã cũ</span><code>ui/co4e_canvas.py</code> 791 · <code>co4e_config_panel.py</code></li>
|
||||
<li><span class="frm">mã cũ</span><code>core/co4e_run_manager.py</code> 331</li>
|
||||
<li><span class="frm">có sẵn</span><code>core/co4e.py</code> — dataclass Workflow/Node/Edge đã có</li>
|
||||
<li><span class="frm done">từ Nam</span><code>FakeConfigRepository</code></li>
|
||||
<li><span class="frm risk">từ Team Hoa</span>DTO <code>ToolPolicyGateway</code> — <b>rủi ro liên team cao nhất</b>, lấy trong hôm nay</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div>
|
||||
<p class="iocap">Output — phải giao</p>
|
||||
<ul class="iolist">
|
||||
<li><code>build_co4e_tab()</code> → <b>cho Nam</b>, trước 28/08</li>
|
||||
<li><code>FakeCo4EWorkflowService</code> → <b>cho cả team</b></li>
|
||||
<li><code>domain/workflows/</code> — DTO chốt ngày đầu</li>
|
||||
<li><code>application/workflows/co4e_workflow_service.py</code></li>
|
||||
<li><code>presentation/co4e/</code> — 5 phần</li>
|
||||
</ul>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
|
||||
<h3>Output bắt buộc với cả ba, mỗi lần đẩy</h3>
|
||||
<div class="scroll">
|
||||
<table>
|
||||
<thead><tr><th>Điều kiện</th><th>Ngưỡng</th><th>Tự kiểm bằng</th></tr></thead>
|
||||
<tbody>
|
||||
<tr><td>File mới sau khi tách</td><td class="mono">≤ 400 dòng</td><td class="mono">wc -l</td></tr>
|
||||
<tr><td><code>domain/</code> và <code>application/</code> import PySide6</td><td class="mono">0</td><td class="mono">grep -r PySide6</td></tr>
|
||||
<tr><td>Test hiện có</td><td class="mono">102 xanh</td><td class="mono">pytest tests -q</td></tr>
|
||||
<tr><td>Credential lộ</td><td class="mono">0</td><td class="mono">python scripts/audit_security.py</td></tr>
|
||||
<tr><td>Checker UI trong phạm vi mình dời</td><td>đã cập nhật</td><td class="mono">python tools/check_<tên>.py</td></tr>
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<h2>Lịch từng ngày</h2>
|
||||
<p class="sub">Ba hàng chạy độc lập. Hàng tô nền là lúc cả ba phải gặp nhau.</p>
|
||||
|
||||
<div class="scroll">
|
||||
<table>
|
||||
<thead>
|
||||
<tr>
|
||||
<th class="day">Ngày</th>
|
||||
<th>N1 · Nam</th>
|
||||
<th>N2 · Hiệp</th>
|
||||
<th>N3 · Lâm</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
<tr>
|
||||
<td class="day">21/08<br><small>T6</small></td>
|
||||
<td class="cl"><b>Mục chung</b> · dựng khung · interface + fake · chốt api_key · CASAN script<small>Merge trước khi hai người kia bắt đầu</small></td>
|
||||
<td class="c1">Chốt schema log 9 trường<small>Giữ nguyên định dạng cũ để 24 chỗ gọi không phải sửa</small></td>
|
||||
<td class="c2">Chốt chữ ký <code>Co4EWorkflowService</code><small>Nộp cho lead để lắp bootstrap sau</small></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td class="day">22–23/08<br><small>T7–CN</small></td>
|
||||
<td class="cl">AtomicJsonFile · ConfigRepository · Typed Settings Facade</td>
|
||||
<td class="c1">CanonicalAuditLogger · gỡ vòng lặp pricing ↔ usage</td>
|
||||
<td class="c2">Co4EWorkflowService — CRUD & validate, test không cần Qt</td>
|
||||
</tr>
|
||||
<tr class="mark">
|
||||
<td class="day">23/08<br><small>17:00</small></td>
|
||||
<td colspan="3"><span class="pill cp">Checkpoint 1</span> 100% DTO và fake xong · <code>pytest</code> xanh · không ai bị chặn</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td class="day">24/08<br><small>T2</small></td>
|
||||
<td class="cl">Tách settings: provider + connector widget</td>
|
||||
<td class="c1">3 tab đầu: overview · sandbox · security events</td>
|
||||
<td class="c2">node_property_panel · agent_list_panel</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td class="day">25/08<br><small>T3</small></td>
|
||||
<td class="cl">Tách settings: routing + general widget</td>
|
||||
<td class="c1">4 tab còn lại: MCP · action logs · agent status · security settings</td>
|
||||
<td class="c2">co4e_canvas_widget — thao tác node</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td class="day">26/08<br><small>T4</small></td>
|
||||
<td class="cl">Chuyển API key sang SecretStore<small>Báo Team Duy trước khi đụng providers/</small></td>
|
||||
<td class="c1">Lắp shell MonitoringTab · query service bản thật</td>
|
||||
<td class="c2">Run control · chat view</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td class="day">27/08<br><small>T5</small></td>
|
||||
<td class="cl">Schema versioning · recovery policy</td>
|
||||
<td class="c1">Ma trận Sandbox · gỡ vòng lặp agent_security</td>
|
||||
<td class="c2">Lắp container Co4ETab · thay fake bằng service thật</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td class="day">28/08<br><small>T6</small></td>
|
||||
<td class="cl"><b>bootstrap.py + tách MainWindow</b><small>Nhận factory của Hiệp và Lâm để lắp</small></td>
|
||||
<td class="c1">Nộp factory · dọn file >400 dòng · cập nhật checker</td>
|
||||
<td class="c2">Nộp factory · dọn file >400 dòng · cập nhật checker</td>
|
||||
</tr>
|
||||
<tr class="mark">
|
||||
<td class="day">28/08<br><small>17:00</small></td>
|
||||
<td colspan="3"><span class="pill cp">Checkpoint 2</span> Tách xong 100% god file · 0 circular import</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td class="day">29/08<br><small>T7</small></td>
|
||||
<td class="cl">Tài liệu Security Policy · integration test Settings</td>
|
||||
<td class="c1">Integration test Monitoring</td>
|
||||
<td class="c2">Integration test luồng Co4E đầu-cuối</td>
|
||||
</tr>
|
||||
<tr class="mark">
|
||||
<td class="day">30/08<br><small>CN 17:00</small></td>
|
||||
<td colspan="3"><span class="pill gate">CASAN Gate</span> <b>Nam chủ trì Check 1</b> — quét toàn bộ config/JSON, phải ra 0 secret plaintext. Hiệp và Lâm sửa ngay phần của mình nếu script bắt được.</td>
|
||||
</tr>
|
||||
<tr class="mark">
|
||||
<td class="day">31/08<br><small>T2 15:00</small></td>
|
||||
<td colspan="3"><span class="pill ship">Bàn giao</span> Fix tồn đọng · cập nhật tài liệu kiến trúc · merge PR cuối · smoke test 5 luồng chính</td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<h2>Nghiệm thu: làm sao biết đã thật sự song song</h2>
|
||||
<p class="sub">Không phải “đã họp xong” mà là chạy được. Ba câu hỏi, trả lời bằng lệnh.</p>
|
||||
|
||||
<div class="scroll">
|
||||
<table>
|
||||
<thead>
|
||||
<tr><th>Câu hỏi</th><th>Cách trả lời</th><th>Khi nào</th></tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
<tr>
|
||||
<td>Hiệp có chạy được khi chưa có config bản thật?</td>
|
||||
<td>Dựng một tab Monitoring, chạy test của nó, <b>không import <code>cowork_local.config</code></b> dòng nào — chỉ dùng <code>FakeConfigRepository</code></td>
|
||||
<td class="mono">21/08</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>Lâm có chạy được khi Team Hoa chưa xong gateway?</td>
|
||||
<td>Test <code>Co4EWorkflowService</code> xanh với <code>FakeToolPolicyGateway</code></td>
|
||||
<td class="mono">23/08</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>Ba người có đụng file nhau không?</td>
|
||||
<td><code>git log --name-only --pretty=%an</code> trên <code>gamma/refactor</code> —
|
||||
không file nào được xuất hiện dưới hai tên khác nhau</td>
|
||||
<td class="mono">mỗi ngày</td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<footer>
|
||||
Nguồn: <code>docs/refactor/plan.md</code>, <code>Refactoring_Checklist.md</code>,
|
||||
<code>Feature_Architecture_Proposal.md</code>. Số dòng code, số lời gọi và baseline test đo
|
||||
trực tiếp trên nhánh <code>main</code> ngày 21/08.
|
||||
Mục chung, cách chia ba nhánh, bảy quy ước và mục nghiệm thu là đề xuất — không có trong
|
||||
tài liệu gốc.
|
||||
</footer>
|
||||
|
||||
</div>
|
||||
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,189 @@
|
||||
# Quyết định của Team Gamma
|
||||
|
||||
Team: **Nam** (nhóm trưởng, nhánh N1) · **Hiệp** (N2) · **Lâm** (N3).
|
||||
|
||||
Ghi ở đây thay vì chôn trong comment, vì cả ba đều ảnh hưởng ra ngoài phạm vi
|
||||
một người.
|
||||
|
||||
| # | Việc | Trạng thái |
|
||||
|---|---|---|
|
||||
| 1 | `provider_conf()` còn trả `api_key` | **Chốt 21/08 — đường A** |
|
||||
| 2 | Số phận 24 checker UI | **Chốt 21/08 — đường A** |
|
||||
| 3 | DTO `ToolPolicyGateway` viết hộ Team Hoa | Đã làm, chờ Hoa xác nhận |
|
||||
|
||||
---
|
||||
|
||||
## Quyết định 1 — `provider_conf()` còn trả `api_key` hay không
|
||||
|
||||
### Vì sao phải quyết trước khi code
|
||||
|
||||
R02-T05 chuyển API key sang Keyring. Câu hỏi là sau khi chuyển, dict do
|
||||
`provider_conf()` trả về **còn chứa `api_key` không**.
|
||||
|
||||
Có 5 nơi đang đọc trực tiếp — đo trên `main` ngày 21/08:
|
||||
|
||||
| Nơi đọc | Thuộc |
|
||||
|---|---|
|
||||
| `providers/anthropic.py:26` | **Team Duy** |
|
||||
| `providers/openai_compat.py:36` | **Team Duy** |
|
||||
| `core/image_gen.py:50` | Team Duy (routing/model) |
|
||||
| `core/ext_connectors.py:98` | Team Hoa |
|
||||
| `ui/ext_connector_dialog.py:87` | Team Gamma |
|
||||
|
||||
Ba trong năm nằm ngoài team. Quyết một mình rồi im lặng là làm vỡ code người khác.
|
||||
|
||||
### Hai đường
|
||||
|
||||
**A. Giữ `api_key` trong dict, `ConfigRepository` tự lấy từ `SecretStore` rồi ghép vào**
|
||||
|
||||
- 5 nơi đọc **không phải sửa dòng nào**
|
||||
- Không cần báo team khác, không cần đồng bộ lịch
|
||||
- Đổi lại: bí mật vẫn đi lang thang trong dict, dễ lọt vào log hoặc màn hình debug
|
||||
- CASAN Check 1 vẫn PASS vì nó quét **file trên đĩa**, không quét bộ nhớ
|
||||
|
||||
**B. Bỏ `api_key` khỏi dict, ai cần thì gọi `secrets.get(provider_key(name))`**
|
||||
|
||||
- Sạch về nguyên tắc: bí mật chỉ xuất hiện đúng chỗ cần
|
||||
- Đổi lại: **5 nơi phải sửa**, 3 trong đó phải chờ team khác xếp lịch
|
||||
- Rủi ro: quên một chỗ thì mất API key lúc chạy thật, mà test có fake nên không bắt được
|
||||
|
||||
### Đề xuất
|
||||
|
||||
**Đường A cho sprint này, đường B ghi vào nợ kỹ thuật.**
|
||||
|
||||
Lý do: mục tiêu của cổng CASAN là *không còn secret nằm trên đĩa*, và đường A
|
||||
đạt được điều đó. Đường B giải quyết thêm chuyện secret trong bộ nhớ — đúng
|
||||
nhưng không phải việc của 10 ngày này, và nó kéo hai team khác vào một thay đổi
|
||||
họ không lên kế hoạch.
|
||||
|
||||
Nếu chọn B thì **phải báo Team Duy và Team Hoa trong hôm nay**, không phải lúc
|
||||
đã sửa xong.
|
||||
|
||||
> **Nam chốt 21/08: đường A.**
|
||||
>
|
||||
> Việc kèm theo: `ConfigRepository` bản thật phải đọc key từ `SecretStore` rồi
|
||||
> ghép vào dict do `provider_conf()` trả về. Năm nơi đọc không đổi một dòng,
|
||||
> nên **không cần báo Duy và Hoa**.
|
||||
>
|
||||
> Nợ kỹ thuật đã ghi: đường B (bỏ `api_key` khỏi dict) để sau sprint này.
|
||||
|
||||
---
|
||||
|
||||
## Quyết định 2 — số phận 24 checker UI
|
||||
|
||||
### Vấn đề
|
||||
|
||||
`tools/check_*.py` là bộ kiểm tra giao diện viết trong 2 tuần vừa rồi, hiện
|
||||
**24 file**. Chúng bám vào đường dẫn cũ:
|
||||
|
||||
| Import | Số chỗ |
|
||||
|---|---|
|
||||
| `cowork_local.config` | 34 |
|
||||
| `cowork_local.app` | 16 |
|
||||
| `cowork_local.state` | 22 |
|
||||
| `cowork_local.ui.*` | ~12 |
|
||||
|
||||
R08 dời hết những module đó sang `presentation/`. Nghĩa là **cả 24 checker chết
|
||||
ngay ngày N1 đụng `config.py`** — và đó là lưới an toàn duy nhất cho phần giao
|
||||
diện, vì `pytest` không kiểm giao diện (90 test hiện tại là logic).
|
||||
|
||||
### Ba đường
|
||||
|
||||
**A. Ai dời file thì cập nhật checker tương ứng, ngay trong PR đó**
|
||||
|
||||
- Giữ được lưới suốt 10 ngày
|
||||
- Tốn thêm ~15% thời gian mỗi PR
|
||||
- Rủi ro: người sửa vội có thể nới lỏng phép kiểm cho nó xanh — đã xảy ra một
|
||||
lần trong quá trình làm UI, khi một checker được sửa thành *không thể đỏ*
|
||||
|
||||
**B. Đóng băng: bỏ khỏi CI, sửa một lượt ngày 31/08**
|
||||
|
||||
- Nhanh nhất trong 10 ngày
|
||||
- Đổi lại: **không có gì canh hồi quy giao diện** suốt cả sprint. Refactor là lúc
|
||||
dễ vỡ giao diện nhất
|
||||
- Rủi ro cuối sprint: sửa 24 file cùng lúc, không ai nhớ cái nào đo gì
|
||||
|
||||
**C. Bỏ hẳn**
|
||||
|
||||
Không khuyến nghị. Vứt đi hai tuần công sức kiểm chứng, và ba tài liệu refactor
|
||||
không có gì thay thế cho phần giao diện.
|
||||
|
||||
### Đề xuất
|
||||
|
||||
**Đường A**, kèm một ràng buộc: PR nào *sửa* checker phải nói rõ trong mô tả
|
||||
**sửa gì và vì sao** — để việc nới lỏng phép kiểm không lọt qua review.
|
||||
|
||||
`tools/check_probes_bite.py` đã có sẵn cơ chế chứng minh checker còn cắn được;
|
||||
chạy nó sau mỗi đợt sửa là bắt được ngay chuyện đó.
|
||||
|
||||
> **Chốt 21/08: đường A** — ai dời file thì cập nhật checker tương ứng ngay
|
||||
> trong PR đó.
|
||||
>
|
||||
> Kèm hai ràng buộc, vì rủi ro của đường A là người sửa vội nới lỏng phép kiểm:
|
||||
>
|
||||
> 1. PR nào *sửa* checker phải nói rõ trong mô tả **sửa gì và vì sao**.
|
||||
> 2. Sửa xong chạy `python tools/check_probes_bite.py` — nó cắm lỗi cố ý vào
|
||||
> code rồi kiểm checker có bắt được không. Chính công cụ này đã từng bắt
|
||||
> được một checker bị sửa thành *không thể đỏ*.
|
||||
>
|
||||
> Không đưa 24 checker vào CI trong sprint này: chúng dựng `MainWindow` thật,
|
||||
> mỗi lần chạy tốn hàng chục giây và thỉnh thoảng sập lúc Qt dọn dẹp. Chạy tay
|
||||
> theo phạm vi mình đụng là đủ.
|
||||
|
||||
---
|
||||
|
||||
## Quyết định 3 — Gamma viết hộ DTO `ToolPolicyGateway` cho Team Hoa
|
||||
|
||||
**Đã làm, chờ Hoa xác nhận.** Ngày: 21/08.
|
||||
|
||||
### Vì sao làm thay
|
||||
|
||||
N3 (Co4E) cần gọi tool nhưng Team Hoa chưa bắt đầu. Ba đường:
|
||||
|
||||
| | Hệ quả |
|
||||
|---|---|
|
||||
| N3 ngồi đợi Hoa | Mất mấy ngày, trái nguyên tắc "không team nào chặn team nào" |
|
||||
| N3 tự phỏng đoán | Phỏng đoán của một người, không ai soi, sửa lại chắc chắn |
|
||||
| **Gamma viết bản đề xuất** | N3 chạy ngay, Hoa có cái cụ thể để duyệt hoặc sửa |
|
||||
|
||||
### Ranh giới không lấn
|
||||
|
||||
Sơ đồ phân hệ trong `plan.md` giao `domain/security/` cho **Team Gamma**, còn
|
||||
`application/conversations/tool_policy_gateway.py` cho **Team Hoa**.
|
||||
|
||||
Nên chia đúng như vậy:
|
||||
|
||||
- **Gamma định nghĩa hình dạng** → `domain/security/tool_policy.py`
|
||||
- **Hoa cài đặt gateway** → `application/conversations/tool_policy_gateway.py`,
|
||||
nối vào `core/mcp_client.py` và tool dựng sẵn
|
||||
|
||||
Không đụng file nào của Hoa.
|
||||
|
||||
### Đã bám vào code đang chạy, không bịa
|
||||
|
||||
| Nguồn | Lấy gì |
|
||||
|---|---|
|
||||
| `core/agent_security.py::SecurityVerdict` | `allowed` · `reason` · `layer` |
|
||||
| `ui/permission_dialog.py` + `chat_panel.py:1312` | trạng thái "hỏi người dùng" |
|
||||
|
||||
Khác biệt duy nhất: gộp thành **một câu trả lời ba trạng thái**
|
||||
(`ALLOW` / `DENY` / `ASK`) thay vì bắt chỗ gọi tự nhớ hỏi hai nơi.
|
||||
|
||||
Hai ràng buộc đưa vào có chủ đích:
|
||||
|
||||
1. `DENY` và `ASK` **bắt buộc có `reason`** — người dùng cần biết vì sao, và
|
||||
`audit_log` cần ghi lại. Thiếu là ném lỗi ngay lúc dựng, không phải lúc chạy.
|
||||
2. `ASK` **không phải** `allowed` — bẫy dễ mắc nhất là coi ASK như ALLOW rồi tool
|
||||
chạy mà chưa ai đồng ý. Có test riêng cho chuyện này.
|
||||
|
||||
### Gửi Hoa cái gì
|
||||
|
||||
> Bên mình viết trước bản đề xuất `ToolPolicyGateway` ở
|
||||
> `domain/security/tool_policy.py` vì N3 cần gọi tool mà bên Hoa chưa bắt đầu —
|
||||
> để N3 khỏi phải tự đoán. Ba kiểu: `ToolCallRequest`, `PolicyDecision`,
|
||||
> `ToolPolicyGateway`. Phần cài đặt vẫn để bên Hoa ở
|
||||
> `application/conversations/tool_policy_gateway.py`, bọn mình không đụng.
|
||||
> Thấy chỗ nào không hợp thì sửa thẳng file đó, đừng tạo kiểu thứ hai. Đổi bây
|
||||
> giờ còn rẻ vì mới mình N3 dùng.
|
||||
|
||||
> Đã gửi Hoa: ☐ — ngày ____ Hoa xác nhận: ☐ đồng ý ☐ có sửa
|
||||
@@ -20,59 +20,22 @@
|
||||
|
||||
---
|
||||
|
||||
## 📊 TIẾN ĐỘ THỰC TẾ — TEAM DUY (cập nhật `2026-08-21 10:55`)
|
||||
|
||||
> [!NOTE]
|
||||
> ### ✅ ĐÃ HOÀN TẤT: 16/16 task của **R01, R03, R04** — đã commit & push lên nhánh `feature/deltateam/refactor-plan`
|
||||
>
|
||||
> | EPIC | Task | Trạng thái |
|
||||
> | :--- | :--- | :--- |
|
||||
> | **R01** Architecture Foundation | T01 → T05 | ✅ 5/5 |
|
||||
> | **R03** Providers & Routing | T01 → T06 | ✅ 6/6 |
|
||||
> | **R04** Agent Runtime & Conversation | T01 → T05 | ✅ 5/5 |
|
||||
>
|
||||
> **Kiểm chứng (chạy thật, không phải ước lượng):**
|
||||
> * `pytest tests/` ➔ **243 pass / 2 fail** trong 44s
|
||||
> * Suite nhanh (`unit + contracts + characterization + routing`) ➔ **218 pass trong 1,16s** (đạt yêu cầu CASAN "A – Automated Tests < 1s cho unit")
|
||||
> * `python scripts/check_imports.py` ➔ **PASS** (0 Qt import trong `domain/`, `application/`)
|
||||
> * Mọi file production mới **< 400 dòng** (lớn nhất: `routing_application_service.py` 353 dòng)
|
||||
> * 2 test fail là **lỗi có sẵn từ trước**, thuộc EPIC **R02**: `config.py` vẫn hardcode `sandbox_pw = "quandh14"` ➔ `tests/test_config_security.py` đỏ
|
||||
>
|
||||
> ### 📍 PHẠM VI TEAM DUY & PHẦN CÒN LẠI
|
||||
> Theo `Feature_Architecture_Proposal.md` (dòng 7) và `DeltaTeam_prompt.md` (dòng 17), Team Duy chủ trì **R01, R03, R04, R08 (phân hệ Chat UI), R10**.
|
||||
> * ✅ **R01, R03, R04** — xong 16/16 task, đã push.
|
||||
> * ⬜ **R08 (R08-T01 ➔ R08-T06)** — chưa bắt đầu: tách `ui/chat_panel.py` (1.795 dòng) thành 6 widget < 400 dòng.
|
||||
> * ⬜ **R10** — làm sau cùng, chờ 3 team hoàn tất.
|
||||
> * **R02 thuộc 🟣 Team Nam** (xem mục EPIC R02 bên dưới) — đây là nguyên nhân 2 test đỏ ở trên, không phải việc của Team Duy.
|
||||
>
|
||||
> ### 📄 BÁO CÁO CHI TIẾT
|
||||
> Xem `docs/refactor/BaoCao_TeamDuy_R01_R03_R04.md` — kết quả từng EPIC, bằng chứng kiểm thử, 3 lỗi thật đã phát hiện, và phạm vi **chưa** kiểm thử.
|
||||
>
|
||||
> ### 📌 CÒN NỢ / CẦN QUYẾT ĐỊNH
|
||||
> 1. `ProviderRegistry` **chưa nối** vào `state.build_provider_for` (vẫn dùng `providers/factory.py`). Nối vào sẽ sửa luôn lỗi: usage của `ollama`/`github_copilot`/`codex` hiện bị ghi nhận nhầm thành `openai_compat` trên Dashboard — nhưng làm vậy sẽ **đổi cách gom dữ liệu lịch sử**.
|
||||
> 2. Mode `fallback` đã hỗ trợ ở config + service nhưng **chưa có trên toggle UI** (thuộc R08).
|
||||
> 3. Đã sửa 2 dòng trong `config.py` (`routing_mode_for` / `set_routing_mode_for`) để dùng chung một bộ từ vựng mode — **cần báo Team Nam** vì file này đang được refactor ở R02.
|
||||
> 4. Circular import `core/model_pricing.py` ↔ `core/usage_tracker.py` **chưa xử lý** (task ngày 28/08).
|
||||
> 5. Việc kế tiếp của Team Duy là **R08 phân hệ Chat UI** (6 widget con), rồi **R10** sau cùng.
|
||||
|
||||
---
|
||||
|
||||
## 📌 PHẦN 1: CHECKLIST CHI TIẾT THEO 10 EPIC (R01 ➔ R10)
|
||||
|
||||
### 🔹 EPIC R01: Architecture Foundation & Characterization (Nền Tảng Kiến Trúc & Test Bảo Vệ)
|
||||
* **Team chịu trách nhiệm**: 🔵 **Team Duy** (Chủ trì ADR & Test Doubles) + Phối hợp cả 3 team
|
||||
* **Mục tiêu**: Khóa DTO, dựng fakes/test doubles chạy offline không phụ thuộc Qt/mạng, thiết lập script chặn vi phạm kiến trúc.
|
||||
|
||||
- [x] **R01-T01 (Team Duy)**: Viết Architecture ADR định rõ ranh giới các tầng ➔ `docs/architecture/ADR-001-layered-architecture.md`
|
||||
*Start: `2026-08-21 09:56` | End: `2026-08-21 10:00`*
|
||||
- [x] **R01-T02 (Team Duy)**: Xây dựng `FakeProvider` và `FakeToolExecutor` chạy offline từ `providers/base.py` ➔ `tests/fakes/fake_provider.py` & `tests/fakes/fake_tool_executor.py`
|
||||
*Start: `2026-08-21 10:00` | End: `2026-08-21 10:02`*
|
||||
- [x] **R01-T03 (Team Duy)**: Viết script quét tĩnh chặn code mới trong `domain/` và `application/` import `PySide6` ➔ `scripts/check_imports.py`
|
||||
*Start: `2026-08-21 09:58` | End: `2026-08-21 10:05`*
|
||||
- [x] **R01-T04 (Team Duy)**: Viết Characterization Tests cho `core/chat_agent.py::run_cowork` ➔ `tests/characterization/test_run_cowork.py`
|
||||
*Start: `2026-08-21 10:02` | End: `2026-08-21 10:04`*
|
||||
- [x] **R01-T05 (Team Duy)**: Lập danh mục và phân loại mã nguồn dormant/dead code ➔ `docs/architecture/dormant-code.md`
|
||||
*Start: `2026-08-21 10:04` | End: `2026-08-21 10:05`*
|
||||
- [ ] **R01-T01 (Team Duy)**: Viết Architecture ADR định rõ ranh giới các tầng ➔ `docs/architecture/ADR-001-layered-architecture.md`
|
||||
*Start: `____-__-__ __:__` | End: `____-__-__ __:__`*
|
||||
- [ ] **R01-T02 (Team Duy)**: Xây dựng `FakeProvider` và `FakeToolExecutor` chạy offline từ `providers/base.py` ➔ `tests/fakes/fake_provider.py` & `tests/fakes/fake_tool_executor.py`
|
||||
*Start: `____-__-__ __:__` | End: `____-__-__ __:__`*
|
||||
- [ ] **R01-T03 (Team Duy)**: Viết script quét tĩnh chặn code mới trong `domain/` và `application/` import `PySide6` ➔ `scripts/check_imports.py`
|
||||
*Start: `____-__-__ __:__` | End: `____-__-__ __:__`*
|
||||
- [ ] **R01-T04 (Team Duy)**: Viết Characterization Tests cho `core/chat_agent.py::run_cowork` ➔ `tests/characterization/test_run_cowork.py`
|
||||
*Start: `____-__-__ __:__` | End: `____-__-__ __:__`*
|
||||
- [ ] **R01-T05 (Team Duy)**: Lập danh mục và phân loại mã nguồn dormant/dead code ➔ `docs/architecture/dormant-code.md`
|
||||
*Start: `____-__-__ __:__` | End: `____-__-__ __:__`*
|
||||
|
||||
---
|
||||
|
||||
@@ -99,18 +62,18 @@
|
||||
* **Team chịu trách nhiệm**: 🔵 **Team Duy** (Chủ trì)
|
||||
* **Mục tiêu**: Hợp nhất logic routing bị phân tán thành `RoutingApplicationService` độc lập Qt; chuẩn hóa catalog nhà cung cấp.
|
||||
|
||||
- [x] **R03-T01 (Team Duy)**: Xây dựng bộ Contract Tests chuẩn hóa cho các Provider từ `providers/base.py` ➔ `tests/contracts/test_providers.py`
|
||||
*Start: `2026-08-21 10:10` | End: `2026-08-21 10:12`*
|
||||
- [x] **R03-T02 (Team Duy)**: Xây dựng `ProviderDescriptor` và `ProviderRegistry` tập trung từ `providers/factory.py` ➔ `domain/models/provider_descriptor.py` & `infrastructure/providers/provider_registry.py`
|
||||
*Start: `2026-08-21 10:06` | End: `2026-08-21 10:10`*
|
||||
- [x] **R03-T03 (Team Duy)**: Xây dựng `RoutingApplicationService` độc lập với Qt từ `core/routing/` ➔ `application/model_routing/routing_application_service.py`
|
||||
*Start: `2026-08-21 10:12` | End: `2026-08-21 10:15`*
|
||||
- [x] **R03-T04 (Team Duy)**: Di chuyển luồng gọi routing từ `ui/chat_panel.py#L638` sang `RoutingApplicationService`
|
||||
*Start: `2026-08-21 10:17` | End: `2026-08-21 10:20`*
|
||||
- [x] **R03-T05 (Team Duy)**: Di chuyển luồng gọi routing từ `ui/co4e_tab.py` và `ui/folder_tab.py` sang `RoutingApplicationService`
|
||||
*Start: `2026-08-21 10:20` | End: `2026-08-21 10:22`*
|
||||
- [x] **R03-T06 (Team Duy)**: Tách logic ghi nhận token usage ra khỏi Provider, chuyển thành `UsageEventSink` ➔ `infrastructure/telemetry/usage_sink.py`
|
||||
*Start: `2026-08-21 10:15` | End: `2026-08-21 10:17`*
|
||||
- [ ] **R03-T01 (Team Duy)**: Xây dựng bộ Contract Tests chuẩn hóa cho các Provider từ `providers/base.py` ➔ `tests/contracts/test_providers.py`
|
||||
*Start: `____-__-__ __:__` | End: `____-__-__ __:__`*
|
||||
- [ ] **R03-T02 (Team Duy)**: Xây dựng `ProviderDescriptor` và `ProviderRegistry` tập trung từ `providers/factory.py` ➔ `domain/models/provider_descriptor.py` & `infrastructure/providers/provider_registry.py`
|
||||
*Start: `____-__-__ __:__` | End: `____-__-__ __:__`*
|
||||
- [ ] **R03-T03 (Team Duy)**: Xây dựng `RoutingApplicationService` độc lập với Qt từ `core/routing/` ➔ `application/model_routing/routing_application_service.py`
|
||||
*Start: `____-__-__ __:__` | End: `____-__-__ __:__`*
|
||||
- [ ] **R03-T04 (Team Duy)**: Di chuyển luồng gọi routing từ `ui/chat_panel.py#L638` sang `RoutingApplicationService`
|
||||
*Start: `____-__-__ __:__` | End: `____-__-__ __:__`*
|
||||
- [ ] **R03-T05 (Team Duy)**: Di chuyển luồng gọi routing từ `ui/co4e_tab.py` và `ui/folder_tab.py` sang `RoutingApplicationService`
|
||||
*Start: `____-__-__ __:__` | End: `____-__-__ __:__`*
|
||||
- [ ] **R03-T06 (Team Duy)**: Tách logic ghi nhận token usage ra khỏi Provider, chuyển thành `UsageEventSink` ➔ `infrastructure/telemetry/usage_sink.py`
|
||||
*Start: `____-__-__ __:__` | End: `____-__-__ __:__`*
|
||||
|
||||
---
|
||||
|
||||
@@ -118,16 +81,16 @@
|
||||
* **Team chịu trách nhiệm**: 🔵 **Team Duy** (Chủ trì)
|
||||
* **Mục tiêu**: Đóng gói input turn chat thành `ConversationExecutionRequest` bất biến, điều phối vòng đời qua `ConversationApplicationService` và phát sinh sự kiện `AgentEvent` có định kiểu.
|
||||
|
||||
- [x] **R04-T01 (Team Duy)**: Định nghĩa immutable dataclass `ConversationExecutionRequest` ➔ `domain/agents/conversation_execution_request.py`
|
||||
*Start: `2026-08-21 10:23` | End: `2026-08-21 10:25`*
|
||||
- [x] **R04-T02 (Team Duy)**: Chuẩn hóa các sự kiện `AgentEvent` (TextChunk, ToolCallStarted, ToolCallResult, Error) ➔ `domain/agents/agent_event.py`
|
||||
*Start: `2026-08-21 10:22` | End: `2026-08-21 10:23`*
|
||||
- [x] **R04-T03 (Team Duy)**: Xây dựng `ConversationApplicationService` điều phối thực thi từ `core/chat_agent.py` ➔ `application/conversations/conversation_application_service.py`
|
||||
*Start: `2026-08-21 10:25` | End: `2026-08-21 10:27`*
|
||||
- [x] **R04-T04 (Team Duy)**: Di chuyển `ui/cowork_tab.py::build_job` sang sử dụng `ConversationExecutionRequest`
|
||||
*Start: `2026-08-21 10:27` | End: `2026-08-21 10:31`*
|
||||
- [x] **R04-T05 (Team Duy)**: Di chuyển `core/task_executors.py` sang dùng chung `ConversationApplicationService`
|
||||
*Start: `2026-08-21 10:28` | End: `2026-08-21 10:30`*
|
||||
- [ ] **R04-T01 (Team Duy)**: Định nghĩa immutable dataclass `ConversationExecutionRequest` ➔ `domain/agents/conversation_execution_request.py`
|
||||
*Start: `____-__-__ __:__` | End: `____-__-__ __:__`*
|
||||
- [ ] **R04-T02 (Team Duy)**: Chuẩn hóa các sự kiện `AgentEvent` (TextChunk, ToolCallStarted, ToolCallResult, Error) ➔ `domain/agents/agent_event.py`
|
||||
*Start: `____-__-__ __:__` | End: `____-__-__ __:__`*
|
||||
- [ ] **R04-T03 (Team Duy)**: Xây dựng `ConversationApplicationService` điều phối thực thi từ `core/chat_agent.py` ➔ `application/conversations/conversation_application_service.py`
|
||||
*Start: `____-__-__ __:__` | End: `____-__-__ __:__`*
|
||||
- [ ] **R04-T04 (Team Duy)**: Di chuyển `ui/cowork_tab.py::build_job` sang sử dụng `ConversationExecutionRequest`
|
||||
*Start: `____-__-__ __:__` | End: `____-__-__ __:__`*
|
||||
- [ ] **R04-T05 (Team Duy)**: Di chuyển `core/task_executors.py` sang dùng chung `ConversationApplicationService`
|
||||
*Start: `____-__-__ __:__` | End: `____-__-__ __:__`*
|
||||
|
||||
---
|
||||
|
||||
@@ -266,16 +229,16 @@
|
||||
|
||||
| Ngày | Task Cần Hoàn Thành | Start Time | End Time | Trạng Thái |
|
||||
| :--- | :--- | :---: | :---: | :---: |
|
||||
| **21/08 (T6)** | Khóa DTO `ConversationExecutionRequest`, `AgentEvent`; Xây dựng `FakeProvider`, `FakeToolExecutor` | `2026-08-21 09:56` | `2026-08-21 10:25` | [x] |
|
||||
| **22-23/08 (T7-CN)** | Chuẩn hóa `ProviderDescriptor`, `ProviderRegistry`; Wrap OpenAI, Anthropic, Ollama, FPT Gateway; Viết Contract Tests | `2026-08-21 10:06` | `2026-08-21 10:12` | [x] ⚠️ registry chưa nối vào `state.build_provider_for` |
|
||||
| **24/08 (T2)** | Xây dựng `RoutingApplicationService` độc lập Qt; Tách `ComposerWidget` & `AttachmentPicker` | `2026-08-21 10:12` | `2026-08-21 10:15` | [~] RoutingApplicationService xong; tách widget thuộc R08 |
|
||||
| **25/08 (T3)** | Xây dựng `ConversationApplicationService`; Tách `ChatHistoryWidget` và bubble renderer | `2026-08-21 10:25` | `2026-08-21 10:27` | [~] Service xong; tách widget thuộc R08 |
|
||||
| **21/08 (T6)** | Khóa DTO `ConversationExecutionRequest`, `AgentEvent`; Xây dựng `FakeProvider`, `FakeToolExecutor` | `____-__-__ __:__` | `____-__-__ __:__` | [ ] |
|
||||
| **22-23/08 (T7-CN)** | Chuẩn hóa `ProviderDescriptor`, `ProviderRegistry`; Wrap OpenAI, Anthropic, Ollama, FPT Gateway; Viết Contract Tests | `____-__-__ __:__` | `____-__-__ __:__` | [ ] |
|
||||
| **24/08 (T2)** | Xây dựng `RoutingApplicationService` độc lập Qt; Tách `ComposerWidget` & `AttachmentPicker` | `____-__-__ __:__` | `____-__-__ __:__` | [ ] |
|
||||
| **25/08 (T3)** | Xây dựng `ConversationApplicationService`; Tách `ChatHistoryWidget` và bubble renderer | `____-__-__ __:__` | `____-__-__ __:__` | [ ] |
|
||||
| **26/08 (T4)** | Nối stream `AgentEvent` sang Chat History; Tách `AudioRecorderWidget` | `____-__-__ __:__` | `____-__-__ __:__` | [ ] |
|
||||
| **27/08 (T5)** | Tách `ChatOutputPanel` & File Watcher; Lắp ráp container `ChatPanel` và `Floating HelpAgent` | `____-__-__ __:__` | `____-__-__ __:__` | [ ] |
|
||||
| **28/08 (T6)** | Xóa copy routing cũ trong `ui/chat_panel.py`; Fix circular import `model_pricing` ↔ `usage_tracker` | `2026-08-21 10:17` | `2026-08-21 10:22` | [~] 3 bản copy routing đã gỡ; circular import chưa xử lý |
|
||||
| **29/08 (T7)** | Viết suite integration test cho toàn bộ luồng Chat (`tests/integration/test_chat_flow.py`) | `2026-08-21 10:35` | `2026-08-21 10:52` | [~] 25 integration test tại `tests/integration/{test_cowork_turn_flow,test_task_executor_flow,test_routing_surfaces}.py` |
|
||||
| **30/08 (CN)** | 🔍 **Chủ trì CASAN Check 3**: Chạy `python scripts/check_imports.py` đảm bảo 0 import `PySide6` trong domain & application | `2026-08-21 09:58` | `2026-08-21 10:05` | [x] PASS |
|
||||
| **31/08 (T2)** | **Chủ trì EPIC R10**: Viết Contributor Recipes, chạy E2E Smoke Test (`tests/e2e/test_smoke.py`) và merge PR cuối cùng | `____-__-__ __:__` | `____-__-__ __:__` | [ ] chờ 3 team hoàn tất |
|
||||
| **28/08 (T6)** | Xóa copy routing cũ trong `ui/chat_panel.py`; Fix circular import `model_pricing` ↔ `usage_tracker` | `____-__-__ __:__` | `____-__-__ __:__` | [ ] |
|
||||
| **29/08 (T7)** | Viết suite integration test cho toàn bộ luồng Chat (`tests/integration/test_chat_flow.py`) | `____-__-__ __:__` | `____-__-__ __:__` | [ ] |
|
||||
| **30/08 (CN)** | 🔍 **Chủ trì CASAN Check 3**: Chạy `python scripts/check_imports.py` đảm bảo 0 import `PySide6` trong domain & application | `____-__-__ __:__` | `____-__-__ __:__` | [ ] |
|
||||
| **31/08 (T2)** | **Chủ trì EPIC R10**: Viết Contributor Recipes, chạy E2E Smoke Test (`tests/e2e/test_smoke.py`) và merge PR cuối cùng | `____-__-__ __:__` | `____-__-__ __:__` | [ ] |
|
||||
|
||||
---
|
||||
|
||||
|
||||
+1
-12
@@ -1,12 +1 @@
|
||||
"""Domain layer - pure Python entities, value objects and events.
|
||||
|
||||
The innermost layer of the 4-tier architecture (see
|
||||
``docs/architecture/ADR-001-layered-architecture.md``). Modules here describe
|
||||
WHAT the application is about - a turn of conversation, a model candidate, an
|
||||
agent event - and depend on nothing but the standard library.
|
||||
|
||||
Hard rule (ADR-001 I1/I2, enforced by ``scripts/check_imports.py``): no imports
|
||||
of PySide6/PyQt, and no imports from ``application/``, ``infrastructure/``,
|
||||
``presentation/`` or the legacy ``core/``/``ui/`` packages. That is what keeps
|
||||
this layer testable in milliseconds and reusable from a headless scheduler.
|
||||
"""
|
||||
"""domain/ — Quy tắc nghiệp vụ thuần. KHÔNG import PySide6, không chạm đĩa/mạng."""
|
||||
|
||||
@@ -1,48 +0,0 @@
|
||||
"""Domain entities for one agent turn: the request snapshot and the typed event
|
||||
stream it produces (EPIC R04)."""
|
||||
|
||||
from .agent_event import (
|
||||
AgentEvent,
|
||||
AssistantDoneEvent,
|
||||
ErrorEvent,
|
||||
HistoryReadyEvent,
|
||||
NoticeEvent,
|
||||
OutputsAddedEvent,
|
||||
OutputsRemovedEvent,
|
||||
PlanUpdatedEvent,
|
||||
ReasoningChunkEvent,
|
||||
TextChunkEvent,
|
||||
ToolCallFinishedEvent,
|
||||
ToolCallStartedEvent,
|
||||
ToolOutputEvent,
|
||||
TurnCompletedEvent,
|
||||
collect_text,
|
||||
event_from_dict,
|
||||
tool_calls,
|
||||
)
|
||||
from .conversation_execution_request import (
|
||||
ConversationExecutionRequest,
|
||||
new_turn_id,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"ConversationExecutionRequest",
|
||||
"new_turn_id",
|
||||
"AgentEvent",
|
||||
"TextChunkEvent",
|
||||
"ReasoningChunkEvent",
|
||||
"AssistantDoneEvent",
|
||||
"PlanUpdatedEvent",
|
||||
"ToolCallStartedEvent",
|
||||
"ToolOutputEvent",
|
||||
"ToolCallFinishedEvent",
|
||||
"OutputsAddedEvent",
|
||||
"OutputsRemovedEvent",
|
||||
"NoticeEvent",
|
||||
"HistoryReadyEvent",
|
||||
"TurnCompletedEvent",
|
||||
"ErrorEvent",
|
||||
"event_from_dict",
|
||||
"collect_text",
|
||||
"tool_calls",
|
||||
]
|
||||
|
||||
@@ -1,370 +0,0 @@
|
||||
"""AgentEvent - the typed event stream one agent turn produces (R04-T02).
|
||||
|
||||
Today the turn engine talks to its caller through untyped dicts::
|
||||
|
||||
emit({"type": "tool_result", "id": tc_id, "name": name,
|
||||
"ok": result.get("ok", False), "output": result.get("output", "")})
|
||||
|
||||
and every consumer re-discovers the vocabulary by reading the producer. There
|
||||
are eleven such shapes across ``core/chat_agent.py``, ``core/code_agent.py`` and
|
||||
``core/task_executors.py``; a consumer that misspells ``"tool_result"`` or reads
|
||||
``"result"`` instead of ``"output"`` fails silently, at runtime, only for the
|
||||
tool path that triggers it.
|
||||
|
||||
This module makes the vocabulary explicit. Each event is a frozen dataclass, so:
|
||||
|
||||
* the set of possible events is enumerable (see :data:`EVENT_TYPES`);
|
||||
* a field name typo is an ``AttributeError`` at the point of use, not a silently
|
||||
missing chat bubble;
|
||||
* an event can cross a thread boundary safely - it cannot be mutated after the
|
||||
producer hands it over, which is exactly what the Qt-signal seam needs.
|
||||
|
||||
Bridging with the legacy dicts is deliberate and two-way: :func:`event_from_dict`
|
||||
adapts what ``run_cowork`` emits today, and :meth:`AgentEvent.to_dict` renders an
|
||||
event back into the legacy shape so existing widgets keep working untouched
|
||||
while the presentation layer migrates screen by screen (EPIC R08).
|
||||
|
||||
Pure domain code: stdlib only, no Qt, no I/O.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Dict, List, Mapping, Optional, Sequence, Tuple
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class AgentEvent:
|
||||
"""Base class for everything a turn can report.
|
||||
|
||||
``type`` is the legacy string tag, kept as a class attribute so the bridge
|
||||
functions can round-trip an event without a separate mapping table.
|
||||
"""
|
||||
|
||||
type: str = field(init=False, default="event")
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
"""Render into the legacy ``emit()`` dict shape."""
|
||||
return {"type": self.type}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Assistant output
|
||||
# --------------------------------------------------------------------------- #
|
||||
@dataclass(frozen=True)
|
||||
class TextChunkEvent(AgentEvent):
|
||||
"""One fragment of the visible answer, as it streams in."""
|
||||
|
||||
delta: str
|
||||
type: str = field(init=False, default="text")
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {"type": self.type, "delta": self.delta}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ReasoningChunkEvent(AgentEvent):
|
||||
"""One fragment of the model's PRIVATE reasoning.
|
||||
|
||||
Drives the "Thinking" indicator only. Consumers must never append this to
|
||||
the answer or persist it into conversation history - keeping it a distinct
|
||||
type is what makes that mistake hard to make by accident.
|
||||
"""
|
||||
|
||||
delta: str
|
||||
type: str = field(init=False, default="reasoning")
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {"type": self.type, "delta": self.delta}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class AssistantDoneEvent(AgentEvent):
|
||||
"""One assistant message finished. A turn with tool calls emits this once
|
||||
per step, not once per turn - see :class:`TurnCompletedEvent`."""
|
||||
|
||||
content: str = ""
|
||||
type: str = field(init=False, default="assistant_done")
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {"type": self.type, "content": self.content}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Planning
|
||||
# --------------------------------------------------------------------------- #
|
||||
@dataclass(frozen=True)
|
||||
class PlanUpdatedEvent(AgentEvent):
|
||||
"""The agent rewrote its plan (the ``update_plan`` tool)."""
|
||||
|
||||
steps: Tuple[Dict[str, Any], ...] = ()
|
||||
type: str = field(init=False, default="plan_set")
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {"type": self.type, "steps": [dict(s) for s in self.steps]}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Tool lifecycle
|
||||
# --------------------------------------------------------------------------- #
|
||||
@dataclass(frozen=True)
|
||||
class ToolCallStartedEvent(AgentEvent):
|
||||
"""A tool call is about to run, with the preview shown to the user.
|
||||
|
||||
Maps the legacy ``tool_proposed`` event. "Proposed" was a misnomer: by the
|
||||
time it is emitted the call is already going to run unless a permission gate
|
||||
rejects it, and the gate reports that as a finished call with ``ok=False``.
|
||||
"""
|
||||
|
||||
call_id: str
|
||||
name: str
|
||||
args: Dict[str, Any] = field(default_factory=dict)
|
||||
preview: Optional[Dict[str, Any]] = None
|
||||
type: str = field(init=False, default="tool_proposed")
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
out: Dict[str, Any] = {"type": self.type, "id": self.call_id,
|
||||
"name": self.name, "args": dict(self.args)}
|
||||
if self.preview is not None:
|
||||
out["preview"] = dict(self.preview)
|
||||
return out
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ToolOutputEvent(AgentEvent):
|
||||
"""A line of live output from a running tool (command stdout, for example)."""
|
||||
|
||||
call_id: str
|
||||
name: str
|
||||
delta: str
|
||||
type: str = field(init=False, default="tool_output")
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {"type": self.type, "id": self.call_id, "name": self.name,
|
||||
"delta": self.delta}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ToolCallFinishedEvent(AgentEvent):
|
||||
"""A tool call ended, successfully or not.
|
||||
|
||||
``ok=False`` covers every failure mode alike - the tool raised, the sandbox
|
||||
blocked it, or the user rejected it at the permission gate - because the
|
||||
consumer's job is the same in all three: show the failure and let the model
|
||||
react to it.
|
||||
"""
|
||||
|
||||
call_id: str
|
||||
name: str
|
||||
ok: bool = False
|
||||
output: str = ""
|
||||
path: str = "" # file the tool wrote, when it wrote one
|
||||
produced: Tuple[str, ...] = () # extra artefacts (e.g. a generator's outputs)
|
||||
type: str = field(init=False, default="tool_result")
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
out: Dict[str, Any] = {"type": self.type, "id": self.call_id, "name": self.name,
|
||||
"ok": self.ok, "output": self.output}
|
||||
if self.path:
|
||||
out["path"] = self.path
|
||||
if self.produced:
|
||||
out["produced"] = list(self.produced)
|
||||
return out
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Output folder
|
||||
# --------------------------------------------------------------------------- #
|
||||
@dataclass(frozen=True)
|
||||
class OutputsAddedEvent(AgentEvent):
|
||||
"""Files appeared in the turn's output folder."""
|
||||
|
||||
paths: Tuple[str, ...] = ()
|
||||
type: str = field(init=False, default="outputs_added")
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {"type": self.type, "paths": list(self.paths)}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class OutputsRemovedEvent(AgentEvent):
|
||||
"""Files were cleaned up from the turn's output folder (intermediates)."""
|
||||
|
||||
paths: Tuple[str, ...] = ()
|
||||
type: str = field(init=False, default="outputs_removed")
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {"type": self.type, "paths": list(self.paths)}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class NoticeEvent(AgentEvent):
|
||||
"""A UI-visible aside that is not part of the model's answer.
|
||||
|
||||
Three producers today, all reachable from a normal turn:
|
||||
``core/agent_security.py`` (a request or command blocked by the security
|
||||
layer), ``core/context_budget.py`` (the conversation was auto-compressed)
|
||||
and the attachment readers (a file that could not be processed, plus live
|
||||
"reading page X/Y" progress).
|
||||
|
||||
``level`` selects how the UI renders it: ``"progress"`` updates the thinking
|
||||
indicator in place, anything else becomes a warning bubble. Dropping these
|
||||
would silently hide security warnings from the user, which is why the type
|
||||
exists rather than being folded into TextChunkEvent.
|
||||
"""
|
||||
|
||||
text: str
|
||||
level: str = "info"
|
||||
type: str = field(init=False, default="notice")
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {"type": self.type, "level": self.level, "text": self.text}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class HistoryReadyEvent(AgentEvent):
|
||||
"""A history session exists for this run and can be opened."""
|
||||
|
||||
session_id: str
|
||||
type: str = field(init=False, default="history_ready")
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {"type": self.type, "session_id": self.session_id}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Turn lifecycle - emitted by the application service, not by the legacy engine
|
||||
# --------------------------------------------------------------------------- #
|
||||
@dataclass(frozen=True)
|
||||
class TurnCompletedEvent(AgentEvent):
|
||||
"""The whole turn finished: no more events will follow.
|
||||
|
||||
New in R04. The legacy engine has no end-of-turn signal at all, so every
|
||||
consumer infers "done" from the worker thread finishing - which is why a
|
||||
cancelled turn and a failed turn look identical to the UI today.
|
||||
"""
|
||||
|
||||
content: str = ""
|
||||
cancelled: bool = False
|
||||
type: str = field(init=False, default="turn_completed")
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {"type": self.type, "content": self.content, "cancelled": self.cancelled}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ErrorEvent(AgentEvent):
|
||||
"""The turn failed. ``recoverable`` marks errors the user can act on
|
||||
(pick another model, shorten the prompt) rather than a hard outage."""
|
||||
|
||||
message: str
|
||||
recoverable: bool = False
|
||||
type: str = field(init=False, default="error")
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {"type": self.type, "message": self.message,
|
||||
"recoverable": self.recoverable}
|
||||
|
||||
|
||||
# The legacy tag -> event class map. Also the authoritative list of what a turn
|
||||
# can emit, which is what makes an exhaustive consumer possible for the first time.
|
||||
EVENT_TYPES: Dict[str, type] = {
|
||||
"text": TextChunkEvent,
|
||||
"reasoning": ReasoningChunkEvent,
|
||||
"assistant_done": AssistantDoneEvent,
|
||||
"plan_set": PlanUpdatedEvent,
|
||||
"tool_proposed": ToolCallStartedEvent,
|
||||
"tool_start": ToolCallStartedEvent,
|
||||
"tool_output": ToolOutputEvent,
|
||||
"tool_result": ToolCallFinishedEvent,
|
||||
"outputs_added": OutputsAddedEvent,
|
||||
"outputs_removed": OutputsRemovedEvent,
|
||||
"notice": NoticeEvent,
|
||||
"history_ready": HistoryReadyEvent,
|
||||
"turn_completed": TurnCompletedEvent,
|
||||
"error": ErrorEvent,
|
||||
}
|
||||
|
||||
|
||||
def event_from_dict(payload: Mapping[str, Any]) -> Optional[AgentEvent]:
|
||||
"""Adapt one legacy ``emit()`` dict into a typed event.
|
||||
|
||||
Returns ``None`` for an unknown tag instead of raising: the legacy engine is
|
||||
still being refactored and may grow an event before this module knows about
|
||||
it. Dropping an unrecognised event degrades the UI by one missing bubble;
|
||||
raising here would abort a turn that had otherwise succeeded.
|
||||
"""
|
||||
kind = str(payload.get("type", ""))
|
||||
cls = EVENT_TYPES.get(kind)
|
||||
if cls is None:
|
||||
return None
|
||||
|
||||
if cls is TextChunkEvent or cls is ReasoningChunkEvent:
|
||||
return cls(delta=str(payload.get("delta", "")))
|
||||
if cls is AssistantDoneEvent:
|
||||
return AssistantDoneEvent(content=str(payload.get("content", "")))
|
||||
if cls is PlanUpdatedEvent:
|
||||
return PlanUpdatedEvent(steps=tuple(payload.get("steps") or ()))
|
||||
if cls is ToolCallStartedEvent:
|
||||
return ToolCallStartedEvent(
|
||||
call_id=str(payload.get("id", "")), name=str(payload.get("name", "")),
|
||||
args=dict(payload.get("args") or {}), preview=payload.get("preview"),
|
||||
)
|
||||
if cls is ToolOutputEvent:
|
||||
return ToolOutputEvent(call_id=str(payload.get("id", "")),
|
||||
name=str(payload.get("name", "")),
|
||||
delta=str(payload.get("delta", "")))
|
||||
if cls is ToolCallFinishedEvent:
|
||||
return ToolCallFinishedEvent(
|
||||
call_id=str(payload.get("id", "")), name=str(payload.get("name", "")),
|
||||
ok=bool(payload.get("ok", False)), output=str(payload.get("output", "")),
|
||||
path=str(payload.get("path", "") or ""),
|
||||
produced=tuple(payload.get("produced") or ()),
|
||||
)
|
||||
if cls is OutputsAddedEvent or cls is OutputsRemovedEvent:
|
||||
return cls(paths=tuple(str(p) for p in (payload.get("paths") or ())))
|
||||
if cls is NoticeEvent:
|
||||
return NoticeEvent(text=str(payload.get("text", "")),
|
||||
level=str(payload.get("level", "info")))
|
||||
if cls is HistoryReadyEvent:
|
||||
return HistoryReadyEvent(session_id=str(payload.get("session_id", "")))
|
||||
if cls is TurnCompletedEvent:
|
||||
return TurnCompletedEvent(content=str(payload.get("content", "")),
|
||||
cancelled=bool(payload.get("cancelled", False)))
|
||||
return ErrorEvent(message=str(payload.get("message", "")),
|
||||
recoverable=bool(payload.get("recoverable", False)))
|
||||
|
||||
|
||||
def collect_text(events: Sequence[AgentEvent]) -> str:
|
||||
"""Join every :class:`TextChunkEvent` - the visible answer, reasoning excluded.
|
||||
|
||||
Provided here so no consumer has to re-derive "which events are the answer",
|
||||
the question the untyped dicts made easy to get wrong.
|
||||
"""
|
||||
return "".join(e.delta for e in events if isinstance(e, TextChunkEvent))
|
||||
|
||||
|
||||
def tool_calls(events: Sequence[AgentEvent]) -> List[ToolCallFinishedEvent]:
|
||||
"""Every finished tool call, in order - for audit views and assertions."""
|
||||
return [e for e in events if isinstance(e, ToolCallFinishedEvent)]
|
||||
|
||||
|
||||
__all__ = [
|
||||
"AgentEvent",
|
||||
"TextChunkEvent",
|
||||
"ReasoningChunkEvent",
|
||||
"AssistantDoneEvent",
|
||||
"PlanUpdatedEvent",
|
||||
"ToolCallStartedEvent",
|
||||
"ToolOutputEvent",
|
||||
"ToolCallFinishedEvent",
|
||||
"OutputsAddedEvent",
|
||||
"OutputsRemovedEvent",
|
||||
"NoticeEvent",
|
||||
"HistoryReadyEvent",
|
||||
"TurnCompletedEvent",
|
||||
"ErrorEvent",
|
||||
"EVENT_TYPES",
|
||||
"event_from_dict",
|
||||
"collect_text",
|
||||
"tool_calls",
|
||||
]
|
||||
@@ -1,192 +0,0 @@
|
||||
"""ConversationExecutionRequest - an immutable snapshot of one turn (R04-T01).
|
||||
|
||||
``ui/cowork_tab.py::build_job`` currently builds a closure that reads widget
|
||||
state from inside the worker thread::
|
||||
|
||||
def job(worker):
|
||||
provider = self.build_provider() # reads combo boxes
|
||||
extra_tools, extra_exec = self.ctx.build_mcp_tools()
|
||||
proj_ctx = project_context_text(load_project(project_id))
|
||||
...
|
||||
|
||||
Everything that closure touches can change while the turn is running: the user
|
||||
can pick another model, switch workspace, or edit the project instructions. The
|
||||
turn then runs on a mixture of old and new state, and which mixture depends on
|
||||
thread timing - the class of bug that reproduces once a week and never in a test.
|
||||
|
||||
This value object is the fix: the presentation layer captures everything a turn
|
||||
needs ON THE UI THREAD, at submit time, into one frozen object. Whatever happens
|
||||
to the widgets afterwards, the turn keeps running on the state the user actually
|
||||
submitted.
|
||||
|
||||
Pure domain code: stdlib only, no Qt, no filesystem access. Paths are held as
|
||||
strings, not ``Path`` objects, so the snapshot stays trivially serialisable -
|
||||
which is what will let a turn be queued, replayed or logged later.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import uuid
|
||||
from dataclasses import dataclass, field, replace
|
||||
from typing import Any, Dict, List, Mapping, Optional, Sequence, Tuple
|
||||
|
||||
# Default tool-use budget for an interactive turn, and the higher ceiling a
|
||||
# run-to-completion step (a Co4E flow step) is allowed. Same numbers
|
||||
# ``core.chat_agent.run_cowork`` defaults to - kept here so the policy is
|
||||
# visible in the request rather than buried in a function signature.
|
||||
DEFAULT_MAX_STEPS = 30
|
||||
DEFAULT_COMPLETION_MAX_STEPS = 200
|
||||
|
||||
|
||||
def new_turn_id() -> str:
|
||||
"""A fresh turn id. Short and random: it only has to be unique within a
|
||||
session's lifetime, and it shows up in log lines humans read."""
|
||||
return uuid.uuid4().hex[:12]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ConversationExecutionRequest:
|
||||
"""Everything one agent turn needs, captured at submit time.
|
||||
|
||||
Attributes:
|
||||
prompt: the user's message for this turn (already assembled, including
|
||||
any attachment text the UI inlined).
|
||||
messages: the full conversation to send, oldest first. Held as a tuple
|
||||
so the snapshot cannot be mutated after capture; use
|
||||
:meth:`message_list` to get the mutable copy the engine expects.
|
||||
output_dir: this turn's OWN folder. Each turn writes into an isolated
|
||||
directory so parallel turns cannot clobber each other's files.
|
||||
session_id: the conversation this turn belongs to.
|
||||
turn_id: unique per turn, for logs and for matching events to a turn.
|
||||
surface: which screen submitted it ("cowork", "co4e", "ai_edit", "task").
|
||||
provider / model: what to run on, already resolved (routing included).
|
||||
Empty ``model`` means "the provider's configured default".
|
||||
title: conversation title, used to name generated files.
|
||||
project_id / project_context: the workspace and its shared instructions,
|
||||
snapshotted so a mid-turn workspace switch cannot change them.
|
||||
agent_role: audit-log attribution for every tool call this turn makes.
|
||||
allowed_tools: permission scope. ``None`` means "all enabled tools";
|
||||
a list restricts the ADVERTISED catalogue, so a read-only step
|
||||
literally cannot be offered a writing tool.
|
||||
max_steps / run_to_completion / completion_max_steps: tool-use budget.
|
||||
enforce_rules: run the security rulebase. Co4E sandboxed runs disable it.
|
||||
confirm_commands: ask before run_command/install_package (permission gate).
|
||||
metadata: free-form extras a caller wants carried along (never
|
||||
interpreted here) - e.g. a scheduled task's id.
|
||||
"""
|
||||
|
||||
prompt: str
|
||||
messages: Tuple[Mapping[str, Any], ...] = ()
|
||||
output_dir: str = ""
|
||||
session_id: str = ""
|
||||
turn_id: str = field(default_factory=new_turn_id)
|
||||
surface: str = "cowork"
|
||||
provider: str = ""
|
||||
model: str = ""
|
||||
title: str = ""
|
||||
project_id: str = ""
|
||||
project_context: str = ""
|
||||
agent_role: str = ""
|
||||
allowed_tools: Optional[Tuple[str, ...]] = None
|
||||
max_steps: int = DEFAULT_MAX_STEPS
|
||||
run_to_completion: bool = False
|
||||
completion_max_steps: int = DEFAULT_COMPLETION_MAX_STEPS
|
||||
enforce_rules: bool = True
|
||||
confirm_commands: bool = False
|
||||
metadata: Mapping[str, Any] = field(default_factory=dict)
|
||||
|
||||
# -- construction helpers ------------------------------------------- #
|
||||
@classmethod
|
||||
def create(cls, prompt: str, messages: Optional[Sequence[Mapping[str, Any]]] = None,
|
||||
**kwargs: Any) -> "ConversationExecutionRequest":
|
||||
"""Build a request from ordinary mutable inputs.
|
||||
|
||||
The messages list is copied element by element, so a later append by the
|
||||
caller (the chat panel keeps appending to its own list) cannot reach
|
||||
inside a request that is already running.
|
||||
"""
|
||||
snapshot = tuple(dict(m) for m in (messages or ()))
|
||||
allowed = kwargs.pop("allowed_tools", None)
|
||||
return cls(prompt=prompt, messages=snapshot,
|
||||
allowed_tools=tuple(allowed) if allowed is not None else None,
|
||||
**kwargs)
|
||||
|
||||
def with_messages(self, messages: Sequence[Mapping[str, Any]]
|
||||
) -> "ConversationExecutionRequest":
|
||||
"""A copy carrying a different message list, everything else unchanged.
|
||||
|
||||
Used when a caller assembles the system prompt or trims history after
|
||||
building the request - it must produce a NEW snapshot rather than mutate
|
||||
the one a turn may already be running on.
|
||||
"""
|
||||
return replace(self, messages=tuple(dict(m) for m in messages))
|
||||
|
||||
def with_model(self, provider: str, model: str) -> "ConversationExecutionRequest":
|
||||
"""A copy pinned to another provider/model - how a routing switch is
|
||||
applied without touching the user's saved settings."""
|
||||
return replace(self, provider=provider, model=model)
|
||||
|
||||
# -- accessors ------------------------------------------------------ #
|
||||
def message_list(self) -> List[Dict[str, Any]]:
|
||||
"""A fresh mutable copy of the messages, for the engine to append to.
|
||||
|
||||
The legacy engine mutates the list it is given (it inserts the system
|
||||
prompt and appends assistant/tool messages). Handing it a copy is what
|
||||
keeps this snapshot immutable in practice and not just by declaration.
|
||||
"""
|
||||
return [dict(m) for m in self.messages]
|
||||
|
||||
@property
|
||||
def effective_max_steps(self) -> int:
|
||||
"""The tool-use ceiling actually in force for this turn."""
|
||||
return self.completion_max_steps if self.run_to_completion else self.max_steps
|
||||
|
||||
@property
|
||||
def has_output_dir(self) -> bool:
|
||||
"""True when this turn may write files."""
|
||||
return bool(self.output_dir)
|
||||
|
||||
def allows_tool(self, name: str) -> bool:
|
||||
"""Whether ``name`` is inside this turn's permission scope.
|
||||
|
||||
``update_plan`` is always allowed: it has no side effects and drives the
|
||||
Plan panel, so scoping it out would silently break the UI rather than
|
||||
restrict a capability.
|
||||
"""
|
||||
if self.allowed_tools is None:
|
||||
return True
|
||||
return name == "update_plan" or name in self.allowed_tools
|
||||
|
||||
def describe(self) -> str:
|
||||
"""Compact one-line identity for log lines."""
|
||||
target = f"{self.provider}/{self.model}" if self.model else self.provider or "default"
|
||||
return f"turn={self.turn_id} surface={self.surface} model={target}"
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
"""JSON-safe projection, for logging a turn or persisting it for replay."""
|
||||
return {
|
||||
"turn_id": self.turn_id,
|
||||
"session_id": self.session_id,
|
||||
"surface": self.surface,
|
||||
"prompt": self.prompt,
|
||||
"message_count": len(self.messages),
|
||||
"output_dir": self.output_dir,
|
||||
"provider": self.provider,
|
||||
"model": self.model,
|
||||
"title": self.title,
|
||||
"project_id": self.project_id,
|
||||
"agent_role": self.agent_role,
|
||||
"allowed_tools": list(self.allowed_tools) if self.allowed_tools is not None else None,
|
||||
"max_steps": self.effective_max_steps,
|
||||
"run_to_completion": self.run_to_completion,
|
||||
"enforce_rules": self.enforce_rules,
|
||||
"confirm_commands": self.confirm_commands,
|
||||
"metadata": dict(self.metadata),
|
||||
}
|
||||
|
||||
|
||||
__all__ = [
|
||||
"ConversationExecutionRequest",
|
||||
"new_turn_id",
|
||||
"DEFAULT_MAX_STEPS",
|
||||
"DEFAULT_COMPLETION_MAX_STEPS",
|
||||
]
|
||||
@@ -1,5 +0,0 @@
|
||||
"""Domain models: provider/model catalogue value objects (EPIC R03)."""
|
||||
|
||||
from .provider_descriptor import ProviderCapability, ProviderDescriptor
|
||||
|
||||
__all__ = ["ProviderDescriptor", "ProviderCapability"]
|
||||
|
||||
@@ -1,171 +0,0 @@
|
||||
"""ProviderDescriptor - the declarative catalogue entry for one model provider (R03-T02).
|
||||
|
||||
Today the knowledge of "what a provider is" is scattered across three places
|
||||
that must be edited together and can silently drift apart:
|
||||
|
||||
* ``providers/factory.py::_REGISTRY`` - name -> implementation class
|
||||
* ``config.py::DEFAULT_CONFIG["providers"]`` - default base_url / model / api_key
|
||||
* ``config.py::PROVIDER_LABELS`` - the human label shown in Settings
|
||||
|
||||
Adding a provider means remembering all three; forgetting one produces a
|
||||
provider that exists but has no label, or a label with no implementation. This
|
||||
value object folds those facts into a single immutable description that the
|
||||
registry (``infrastructure/providers/provider_registry.py``) and the UI can both
|
||||
read, so a new provider is declared once.
|
||||
|
||||
Pure domain code: stdlib only, no Qt, no network, no config access. It describes
|
||||
a provider; building one is infrastructure's job.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from enum import Enum
|
||||
from typing import Any, Dict, FrozenSet, List, Mapping, Optional, Tuple
|
||||
|
||||
|
||||
class ProviderCapability(str, Enum):
|
||||
"""What a provider can do, as advertised by its descriptor.
|
||||
|
||||
Kept as a closed enum rather than free-form strings so a typo
|
||||
(``"vison"``) fails at import time instead of silently disabling a feature
|
||||
at runtime. Inherits ``str`` so existing dict/JSON code that compares against
|
||||
plain strings keeps working during the migration.
|
||||
"""
|
||||
|
||||
STREAMING = "streaming" # can stream answer fragments through on_text
|
||||
TOOLS = "tools" # can be given a ToolSpec catalogue and call tools
|
||||
VISION = "vision" # accepts image content blocks (see providers/base.py)
|
||||
REASONING = "reasoning" # emits a separate private "thinking" stream
|
||||
MODEL_LISTING = "model_listing" # list_models() returns a real catalogue
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ProviderDescriptor:
|
||||
"""An immutable description of one provider the app can talk to.
|
||||
|
||||
Attributes:
|
||||
id: the config key, e.g. ``"openai_compat"``. Also the ``provider`` half
|
||||
of a routing candidate key (``provider/model_id``).
|
||||
label: human-readable name for Settings and the model picker.
|
||||
protocol: which wire format this provider speaks. Several ids share one
|
||||
protocol - ``ollama``, ``github_copilot`` and ``codex`` are all
|
||||
OpenAI-compatible endpoints - which is exactly why protocol and id
|
||||
must be separate fields.
|
||||
default_model: the model used when the user has not chosen one.
|
||||
capabilities: what the provider supports (see :class:`ProviderCapability`).
|
||||
requires_api_key: whether an empty ``api_key`` makes it unusable.
|
||||
requires_base_url: whether an empty ``base_url`` makes it unusable.
|
||||
local: True when the endpoint runs on the user's own machine. Routing
|
||||
treats local models as zero-cost, and the security layer treats them
|
||||
as not leaving the machine, so this is a real behavioural flag and
|
||||
not just documentation.
|
||||
notes: free-form remark shown in Settings (e.g. "paste a Copilot token").
|
||||
"""
|
||||
|
||||
id: str
|
||||
label: str
|
||||
protocol: str
|
||||
default_model: str = ""
|
||||
capabilities: FrozenSet[ProviderCapability] = field(default_factory=frozenset)
|
||||
requires_api_key: bool = True
|
||||
requires_base_url: bool = True
|
||||
local: bool = False
|
||||
notes: str = ""
|
||||
|
||||
# -- capability queries ---------------------------------------------- #
|
||||
def supports(self, capability: ProviderCapability) -> bool:
|
||||
"""True when this provider advertises ``capability``."""
|
||||
return capability in self.capabilities
|
||||
|
||||
@property
|
||||
def supports_vision(self) -> bool:
|
||||
"""Mirrors ``providers.base.Provider.supports_vision`` so callers can ask
|
||||
the descriptor (no instance, no network) before building a provider."""
|
||||
return self.supports(ProviderCapability.VISION)
|
||||
|
||||
@property
|
||||
def supports_tools(self) -> bool:
|
||||
"""True when this provider can run an agent turn with tools. A provider
|
||||
without it can still chat, but must never be routed a tool-using task."""
|
||||
return self.supports(ProviderCapability.TOOLS)
|
||||
|
||||
def capability_names(self) -> List[str]:
|
||||
"""Capabilities as sorted plain strings - the shape the routing layer's
|
||||
``required_capabilities`` filter and the assessment store both use."""
|
||||
return sorted(c.value for c in self.capabilities)
|
||||
|
||||
# -- configuration validation ---------------------------------------- #
|
||||
def missing_settings(self, conf: Mapping[str, Any]) -> List[str]:
|
||||
"""Which required config keys are absent or blank in ``conf``.
|
||||
|
||||
Returned as a list (not a bool) so Settings can tell the user exactly
|
||||
what to fill in, instead of a generic "not configured". A provider that
|
||||
needs nothing returns an empty list.
|
||||
"""
|
||||
missing: List[str] = []
|
||||
if self.requires_api_key and not str(conf.get("api_key", "") or "").strip():
|
||||
missing.append("api_key")
|
||||
if self.requires_base_url and not str(conf.get("base_url", "") or "").strip():
|
||||
missing.append("base_url")
|
||||
return missing
|
||||
|
||||
def is_configured(self, conf: Mapping[str, Any]) -> bool:
|
||||
"""True when ``conf`` carries everything this provider needs to run."""
|
||||
return not self.missing_settings(conf)
|
||||
|
||||
def resolve_model(self, conf: Optional[Mapping[str, Any]] = None,
|
||||
requested: str = "") -> str:
|
||||
"""Pick the model id for a call: explicit request, else configured, else
|
||||
this descriptor's default.
|
||||
|
||||
Centralised here because the same three-step fallback is currently
|
||||
re-implemented at every call site (chat panel, Co4E, AI-edit, scheduler),
|
||||
and each of them gets the precedence subtly different.
|
||||
"""
|
||||
if requested:
|
||||
return requested
|
||||
configured = str((conf or {}).get("model", "") or "").strip()
|
||||
return configured or self.default_model
|
||||
|
||||
def describe(self, conf: Optional[Mapping[str, Any]] = None) -> str:
|
||||
"""One-line summary for logs and the Settings row, e.g.
|
||||
``"anthropic:claude-sonnet-4-6 (Anthropic Claude)"``."""
|
||||
return f"{self.id}:{self.resolve_model(conf)} ({self.label})"
|
||||
|
||||
def candidate_key(self, model_id: str) -> str:
|
||||
"""The ``provider/model_id`` identity the routing layer keys on.
|
||||
|
||||
Defined here so the domain owns the format; ``core.routing.models`` has
|
||||
its own ``candidate_key()`` helper producing the identical string, and
|
||||
keeping them equal is what lets the new registry and the existing
|
||||
assessment store share one keyspace during the migration.
|
||||
"""
|
||||
return f"{self.id}/{model_id}"
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
"""JSON-safe projection, for persisting a catalogue snapshot or sending
|
||||
the descriptor to a UI layer that must not import domain types."""
|
||||
return {
|
||||
"id": self.id,
|
||||
"label": self.label,
|
||||
"protocol": self.protocol,
|
||||
"default_model": self.default_model,
|
||||
"capabilities": self.capability_names(),
|
||||
"requires_api_key": self.requires_api_key,
|
||||
"requires_base_url": self.requires_base_url,
|
||||
"local": self.local,
|
||||
"notes": self.notes,
|
||||
}
|
||||
|
||||
|
||||
def split_candidate_key(key: str) -> Tuple[str, str]:
|
||||
"""Inverse of :meth:`ProviderDescriptor.candidate_key`.
|
||||
|
||||
Splits on the FIRST ``/`` only: some gateways expose model ids that contain
|
||||
a slash (``org/model``), and splitting on the last one would corrupt them.
|
||||
"""
|
||||
provider, _, model_id = key.partition("/")
|
||||
return provider, model_id
|
||||
|
||||
|
||||
__all__ = ["ProviderCapability", "ProviderDescriptor", "split_candidate_key"]
|
||||
@@ -0,0 +1,110 @@
|
||||
"""Cổng chính sách cho lời gọi tool — hình dạng dữ liệu, chưa phải cài đặt.
|
||||
|
||||
BẢN ĐỀ XUẤT, chờ Team Hoa xác nhận
|
||||
==================================
|
||||
Sơ đồ phân hệ trong ``plan.md`` giao ``domain/security/`` cho Team Gamma và
|
||||
``application/conversations/tool_policy_gateway.py`` cho Team Hoa. Nên Gamma
|
||||
định nghĩa *hình dạng*, Hoa *cài đặt*.
|
||||
|
||||
Viết trước vì N3 (Co4E) cần gọi tool và Team Hoa chưa bắt đầu. Không có nó thì
|
||||
N3 phải tự phỏng đoán rồi sửa lại sau — mà phỏng đoán của một người thì tệ hơn
|
||||
một đề xuất viết ra để cả hai bên soi.
|
||||
|
||||
Nếu Hoa thấy khác, sửa file này chứ đừng đẻ kiểu thứ hai. Đổi sớm rẻ hơn đổi
|
||||
muộn: hiện chỉ N3 dùng.
|
||||
|
||||
Mô hình bám theo code đang chạy, không bịa:
|
||||
* ``core/agent_security.py::SecurityVerdict`` — allowed / reason / layer
|
||||
* ``ui/permission_dialog.py`` — hộp thoại hỏi người dùng khi
|
||||
``ctx.project_confirm_commands()`` bật (``ui/chat_panel.py:1312``)
|
||||
|
||||
Điểm khác biệt duy nhất so với hôm nay: gộp hai thứ đó thành **một câu trả lời
|
||||
ba trạng thái**, thay vì code gọi phải tự nhớ hỏi cả hai nơi.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from enum import Enum
|
||||
from typing import Any, Dict, Protocol, runtime_checkable
|
||||
|
||||
|
||||
class PolicyOutcome(str, Enum):
|
||||
"""Ba trạng thái. ``ASK`` là thứ hệ thống hiện tại đã có (hộp thoại xin
|
||||
phép) nhưng chưa được coi là một kết quả chính thức."""
|
||||
|
||||
ALLOW = "allow"
|
||||
DENY = "deny"
|
||||
ASK = "ask"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ToolCallRequest:
|
||||
"""Một lời gọi tool đang chờ được duyệt.
|
||||
|
||||
``surface`` cho biết chỗ phát sinh — ``"cowork"``, ``"code"``, ``"co4e"``,
|
||||
``"task"``. Chính sách khác nhau theo màn: Co4E chạy nền nên không thể bật
|
||||
hộp thoại hỏi giữa chừng như Cowork.
|
||||
"""
|
||||
|
||||
name: str
|
||||
arguments: Dict[str, Any] = field(default_factory=dict)
|
||||
surface: str = "cowork"
|
||||
project_id: str = ""
|
||||
#: True nếu tool đến từ MCP server ngoài, False nếu là tool dựng sẵn.
|
||||
external: bool = False
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PolicyDecision:
|
||||
"""Câu trả lời của cổng.
|
||||
|
||||
``reason`` bắt buộc có khi DENY hoặc ASK — người dùng phải biết vì sao bị
|
||||
chặn, và ``core/audit_log.py`` cần nó để ghi lại.
|
||||
|
||||
``layer`` giữ đúng từ vựng của ``SecurityVerdict``: ``"prompt"`` |
|
||||
``"attachment"`` | ``"command"``, cộng thêm ``"policy"`` cho quyết định của
|
||||
chính cổng này.
|
||||
"""
|
||||
|
||||
outcome: PolicyOutcome
|
||||
reason: str = ""
|
||||
layer: str = "policy"
|
||||
|
||||
@property
|
||||
def allowed(self) -> bool:
|
||||
"""Tương thích với chỗ đang đọc ``SecurityVerdict.allowed``.
|
||||
|
||||
Chú ý: ``ASK`` KHÔNG phải allowed — còn phải hỏi người dùng đã.
|
||||
"""
|
||||
return self.outcome is PolicyOutcome.ALLOW
|
||||
|
||||
def __post_init__(self):
|
||||
if self.outcome is not PolicyOutcome.ALLOW and not self.reason:
|
||||
raise ValueError("DENY và ASK bắt buộc có reason — người dùng và "
|
||||
"audit log đều cần biết vì sao")
|
||||
|
||||
|
||||
def allow() -> PolicyDecision:
|
||||
return PolicyDecision(PolicyOutcome.ALLOW)
|
||||
|
||||
|
||||
def deny(reason: str, layer: str = "policy") -> PolicyDecision:
|
||||
return PolicyDecision(PolicyOutcome.DENY, reason, layer)
|
||||
|
||||
|
||||
def ask(reason: str, layer: str = "policy") -> PolicyDecision:
|
||||
return PolicyDecision(PolicyOutcome.ASK, reason, layer)
|
||||
|
||||
|
||||
@runtime_checkable
|
||||
class ToolPolicyGateway(Protocol):
|
||||
"""Hỏi trước khi chạy tool. Cài đặt thật: Team Hoa (R07, hạn 29/08)."""
|
||||
|
||||
def check(self, request: ToolCallRequest) -> PolicyDecision:
|
||||
"""Được chạy tool này không.
|
||||
|
||||
KHÔNG được tự bật hộp thoại bên trong — cổng chỉ *trả lời*, còn hỏi ai
|
||||
và hỏi thế nào là việc của tầng giao diện. Có vậy thì Co4E chạy nền mới
|
||||
dùng chung cổng được với Cowork chạy tương tác.
|
||||
"""
|
||||
...
|
||||
@@ -1,7 +1 @@
|
||||
"""Infrastructure layer - adapters to the outside world.
|
||||
|
||||
Concrete implementations of what the inner layers only describe: HTTP calls to
|
||||
model gateways, the OS keyring, the filesystem, subprocesses, telemetry sinks.
|
||||
May import ``domain/`` (to speak its types) and third-party libraries, but never
|
||||
``presentation/``/``ui/``.
|
||||
"""
|
||||
"""infrastructure/ — Chạm thế giới thật: file, keyring, HTTP, tiến trình. Cài đặt interface."""
|
||||
|
||||
@@ -0,0 +1,104 @@
|
||||
"""Cấu hình ứng dụng — interface, chưa phải cài đặt.
|
||||
|
||||
Hợp đồng số 2 của mục chung. Đây là thứ gỡ chốt lớn nhất: **156 lời gọi
|
||||
``ctx.config.*`` nằm rải trong 29 file**, nên nếu N2 và N3 phải đợi
|
||||
``ConfigRepository`` bản thật (R02-T02, hạn 23/08) thì hai người mất mấy ngày
|
||||
đầu ngồi không.
|
||||
|
||||
Danh sách thuộc tính dưới đây không bịa ra: đếm trực tiếp chỗ đang gọi trong
|
||||
``core/``, ``ui/``, ``providers/`` và ``app.py`` rồi lấy những cái được dùng
|
||||
thật, xếp theo số lần gọi.
|
||||
|
||||
Một chỗ cố ý KHÔNG đưa vào: ``config.data`` (36 lần gọi, nhiều nhất). Đó là
|
||||
đống dict thô — cho nó vào interface là bê nguyên vấn đề cũ sang kiến trúc mới.
|
||||
Ai đang cần ``data`` thì mở issue để bổ sung một thuộc tính có kiểu rõ ràng.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Protocol, runtime_checkable
|
||||
|
||||
|
||||
@runtime_checkable
|
||||
class ConfigRepository(Protocol):
|
||||
"""Đọc/ghi cấu hình. Cài đặt thật dùng ``AtomicJsonFile`` (R02-T01/T02)."""
|
||||
|
||||
# ---- provider ------------------------------------------------------
|
||||
@property
|
||||
def active_provider(self) -> str:
|
||||
"""Tên provider đang chọn (24 lời gọi)."""
|
||||
...
|
||||
|
||||
def set_active_provider(self, name: str) -> None:
|
||||
...
|
||||
|
||||
def provider_conf(self, name: str | None = None) -> Dict[str, Any]:
|
||||
"""Cấu hình của một provider (9 lời gọi).
|
||||
|
||||
CHÚ Ý — điểm còn bỏ ngỏ, xem ``docs/refactor/GammaTeam_decisions.md``:
|
||||
dict này còn chứa ``api_key`` hay không là quyết định chưa chốt. Có 5
|
||||
nơi đang đọc trực tiếp, 3 trong số đó thuộc ``providers/`` của Team Duy.
|
||||
"""
|
||||
...
|
||||
|
||||
# ---- đường dẫn -----------------------------------------------------
|
||||
@property
|
||||
def shared_dir(self) -> str:
|
||||
"""Thư mục dùng chung cho telemetry nhiều máy (10 lời gọi)."""
|
||||
...
|
||||
|
||||
def history_dir(self) -> Path:
|
||||
"""Thư mục lịch sử chat của project đang chọn (7 lời gọi)."""
|
||||
...
|
||||
|
||||
def cowork_output_dir(self) -> Path:
|
||||
"""Thư mục Cowork ghi kết quả ra (6 lời gọi)."""
|
||||
...
|
||||
|
||||
# ---- giao diện -----------------------------------------------------
|
||||
@property
|
||||
def theme(self) -> str:
|
||||
"""``"dark"`` | ``"light"`` | ``"system"`` (8 lời gọi)."""
|
||||
...
|
||||
|
||||
def set_theme(self, value: str) -> None:
|
||||
...
|
||||
|
||||
@property
|
||||
def language(self) -> str:
|
||||
"""``"vi"`` | ``"en"`` | ``"ja"`` (4 lời gọi)."""
|
||||
...
|
||||
|
||||
def set_language(self, value: str) -> None:
|
||||
...
|
||||
|
||||
# ---- các nhóm cấu hình còn lại -------------------------------------
|
||||
@property
|
||||
def routing(self) -> Dict[str, Any]:
|
||||
"""Cấu hình định tuyến model (7 lời gọi)."""
|
||||
...
|
||||
|
||||
@property
|
||||
def auth(self) -> Dict[str, Any]:
|
||||
"""Cấu hình đăng nhập (6 lời gọi)."""
|
||||
...
|
||||
|
||||
@property
|
||||
def agent_security(self) -> Dict[str, Any]:
|
||||
"""Chính sách an toàn cho agent (5 lời gọi)."""
|
||||
...
|
||||
|
||||
@property
|
||||
def tools_disabled(self) -> list[str]:
|
||||
"""Tool bị tắt (2 lời gọi)."""
|
||||
...
|
||||
|
||||
def set_tool_enabled(self, name: str, enabled: bool) -> None:
|
||||
...
|
||||
|
||||
# ---- ghi ------------------------------------------------------------
|
||||
def save(self) -> None:
|
||||
"""Ghi xuống đĩa. Bản thật ghi atomic — tạm + fsync + thay thế —
|
||||
nên tắt máy giữa chừng không làm hỏng file (R02-T01).
|
||||
"""
|
||||
...
|
||||
@@ -0,0 +1,197 @@
|
||||
"""ConfigRepository chạy trên file JSON — R02-T02.
|
||||
|
||||
Thay cho ``config.py::AppConfig``. Hai khác biệt duy nhất về hành vi, cả hai
|
||||
đều là thứ ta muốn:
|
||||
|
||||
1. Ghi qua :class:`AtomicJsonFile` — mất điện giữa lúc lưu không còn làm hỏng
|
||||
cấu hình (R02-T01).
|
||||
2. API key đọc từ :class:`SecretStore` rồi **ghép vào** dict do
|
||||
``provider_conf()`` trả về — đúng đường A đã chốt 21/08
|
||||
(``docs/refactor/GammaTeam_decisions.md``). Nhờ vậy 5 nơi đang đọc
|
||||
``conf["api_key"]`` không phải sửa dòng nào, trong đó 3 nơi thuộc Team Duy.
|
||||
|
||||
Mọi thứ còn lại giữ nguyên có chủ đích: trộn sâu với mặc định, đọc biến môi
|
||||
trường, ``ms365.unlocked`` không bao giờ chạm đĩa. Đây là refactor — hành vi
|
||||
nhìn từ ngoài phải y hệt.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict
|
||||
|
||||
from ..persistence.json.atomic_json_file import AtomicJsonFile
|
||||
from ..secrets.secret_store import SecretStore, provider_key
|
||||
from .schema_migration import CURRENT_VERSION, migrate
|
||||
|
||||
|
||||
class JsonConfigRepository:
|
||||
"""Cấu hình đọc/ghi từ một file JSON, bí mật để trong ``SecretStore``.
|
||||
|
||||
``secrets`` để None nghĩa là không có kho bí mật — mọi thứ vẫn chạy, chỉ
|
||||
là ``api_key`` lấy nguyên từ file như trước. Cần vậy để chuyển dần
|
||||
(R02-T05) chứ không phải đổi một phát cả app.
|
||||
"""
|
||||
|
||||
def __init__(self, path: Path, *, secrets: SecretStore | None = None,
|
||||
defaults: Dict[str, Any] | None = None,
|
||||
env_overrides=None):
|
||||
self._file = AtomicJsonFile(path)
|
||||
self._secrets = secrets
|
||||
# Lấy thẳng từ config.py để hai bên không lệch nhau trong lúc chuyển.
|
||||
if defaults is None or env_overrides is None:
|
||||
from ... import config as legacy
|
||||
defaults = defaults if defaults is not None else legacy.DEFAULT_CONFIG
|
||||
env_overrides = env_overrides or legacy._apply_env_overrides
|
||||
self._defaults = defaults
|
||||
self._env_overrides = env_overrides
|
||||
self.data: Dict[str, Any] = self._load()
|
||||
|
||||
# ---- nạp ------------------------------------------------------------
|
||||
def _load(self) -> Dict[str, Any]:
|
||||
merged = copy.deepcopy(self._defaults)
|
||||
stored = self._file.read(default=None)
|
||||
if isinstance(stored, dict):
|
||||
# Nâng cấp TRƯỚC khi trộn với mặc định: bước v1→v2 gỡ api_key khỏi
|
||||
# đĩa, mà mặc định thì không có khoá nào để gỡ.
|
||||
stored, changed = migrate(stored, secrets=self._secrets,
|
||||
path=self._file.path)
|
||||
merged = _deep_merge(merged, stored)
|
||||
if changed:
|
||||
self.data = merged
|
||||
self.save() # ghi ngay, để lần sau khỏi chuyển lại
|
||||
merged = self._env_overrides(merged)
|
||||
# Trạng thái mở khoá ms365 chỉ tồn tại lúc chạy — mỗi lần mở app đều
|
||||
# bắt đầu ở trạng thái khoá, không tin giá trị đọc từ đĩa.
|
||||
merged.setdefault("ms365", {})["unlocked"] = False
|
||||
return merged
|
||||
|
||||
def reload(self) -> None:
|
||||
self.data = self._load()
|
||||
|
||||
# ---- provider --------------------------------------------------------
|
||||
@property
|
||||
def active_provider(self) -> str:
|
||||
return self.data.get("active_provider", "")
|
||||
|
||||
def set_active_provider(self, name: str) -> None:
|
||||
self.data["active_provider"] = name
|
||||
|
||||
def provider_conf(self, name: str | None = None) -> Dict[str, Any]:
|
||||
"""Cấu hình provider, có sẵn ``api_key``.
|
||||
|
||||
Trả về BẢN SAO: chỗ gọi sửa dict này thì không được âm thầm ghi ngược
|
||||
vào cấu hình — và quan trọng hơn, khoá vừa ghép vào không được lẫn
|
||||
ngược vào ``self.data`` rồi theo ``save()`` xuống đĩa.
|
||||
"""
|
||||
name = name or self.active_provider
|
||||
conf = dict(self.data.get("providers", {}).get(name, {}))
|
||||
if self._secrets is not None:
|
||||
stored = self._secrets.get(provider_key(name))
|
||||
if stored:
|
||||
conf["api_key"] = stored
|
||||
return conf
|
||||
|
||||
def set_api_key(self, name: str, value: str) -> None:
|
||||
"""Lưu khoá vào kho bí mật, và xoá khỏi cấu hình trên đĩa.
|
||||
|
||||
Đây là nửa còn lại của đường A: dict *đọc ra* vẫn có ``api_key``,
|
||||
nhưng file JSON *trên đĩa* thì không — điều kiện để qua CASAN Check 1.
|
||||
"""
|
||||
if self._secrets is not None:
|
||||
self._secrets.set(provider_key(name), value)
|
||||
self.data.setdefault("providers", {}).setdefault(name, {})["api_key"] = ""
|
||||
else:
|
||||
self.data.setdefault("providers", {}).setdefault(name, {})["api_key"] = value
|
||||
|
||||
# ---- đường dẫn -------------------------------------------------------
|
||||
@property
|
||||
def shared_dir(self) -> str:
|
||||
return self.data.get("shared_dir", "")
|
||||
|
||||
def history_dir(self) -> Path:
|
||||
rt = self.data.get("_project_history_dir")
|
||||
if rt:
|
||||
return Path(rt)
|
||||
custom = (self.data.get("history", {}).get("custom_dir") or "").strip()
|
||||
if custom:
|
||||
return Path(custom).expanduser()
|
||||
from ...config import CONFIG_DIR
|
||||
return CONFIG_DIR / "history"
|
||||
|
||||
def cowork_output_dir(self) -> Path:
|
||||
custom = (self.data.get("cowork", {}).get("output_dir") or "").strip()
|
||||
if custom:
|
||||
return Path(custom).expanduser()
|
||||
from ... import paths
|
||||
from ...config import CONFIG_DIR
|
||||
root = paths.primary_onedrive_root()
|
||||
if root is not None:
|
||||
return root / "CoworkLocal" / "output"
|
||||
return CONFIG_DIR / "output" / "cowork"
|
||||
|
||||
# ---- giao diện -------------------------------------------------------
|
||||
@property
|
||||
def theme(self) -> str:
|
||||
return self.data.get("theme", "dark")
|
||||
|
||||
def set_theme(self, value: str) -> None:
|
||||
self.data["theme"] = value
|
||||
|
||||
@property
|
||||
def language(self) -> str:
|
||||
return self.data.get("language", "vi")
|
||||
|
||||
def set_language(self, value: str) -> None:
|
||||
self.data["language"] = value
|
||||
|
||||
# ---- nhóm cấu hình ---------------------------------------------------
|
||||
@property
|
||||
def routing(self) -> Dict[str, Any]:
|
||||
return self.data.setdefault("routing", {})
|
||||
|
||||
@property
|
||||
def auth(self) -> Dict[str, Any]:
|
||||
return self.data.setdefault("auth", {})
|
||||
|
||||
@property
|
||||
def agent_security(self) -> Dict[str, Any]:
|
||||
return self.data.setdefault("agent_security", {})
|
||||
|
||||
@property
|
||||
def tools_disabled(self) -> list[str]:
|
||||
return list(self.data.get("tools_disabled", []))
|
||||
|
||||
def set_tool_enabled(self, name: str, enabled: bool) -> None:
|
||||
disabled = list(self.data.get("tools_disabled", []))
|
||||
if enabled:
|
||||
disabled = [t for t in disabled if t != name]
|
||||
elif name not in disabled:
|
||||
disabled.append(name)
|
||||
self.data["tools_disabled"] = disabled
|
||||
|
||||
# ---- ghi -------------------------------------------------------------
|
||||
def save(self) -> None:
|
||||
"""Ghi nguyên tử. Không bao giờ để lộ trạng thái mở khoá ms365."""
|
||||
to_write = self.data
|
||||
if self.data.get("ms365", {}).get("unlocked"):
|
||||
to_write = copy.deepcopy(self.data)
|
||||
to_write["ms365"]["unlocked"] = False
|
||||
to_write.pop("_project_history_dir", None)
|
||||
to_write["schema_version"] = CURRENT_VERSION
|
||||
self._file.write(to_write)
|
||||
|
||||
|
||||
def _deep_merge(base: Dict[str, Any], override: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Trộn sâu — giống hệt ``config.py::_deep_merge``.
|
||||
|
||||
Không import lại từ đó vì file này phải sống được sau khi ``config.py``
|
||||
biến mất; giữ bản sao 6 dòng còn hơn giữ một sợi dây phụ thuộc.
|
||||
"""
|
||||
out = copy.deepcopy(base)
|
||||
for key, value in (override or {}).items():
|
||||
if isinstance(value, dict) and isinstance(out.get(key), dict):
|
||||
out[key] = _deep_merge(out[key], value)
|
||||
else:
|
||||
out[key] = value
|
||||
return out
|
||||
@@ -0,0 +1,134 @@
|
||||
"""Đánh số phiên bản và chuyển đổi cấu hình — R02-T06.
|
||||
|
||||
Hôm nay ``config.json`` không có số phiên bản. Nghĩa là không có cách nào biết
|
||||
file trên đĩa thuộc thời nào, và mọi thay đổi hình dạng phải xử lý bằng cách
|
||||
đoán — ``config.py::_migrate_connectors()`` chính là một ví dụ: nó đoán "có
|
||||
khoá ``office`` nghĩa là file cũ".
|
||||
|
||||
Ở đây đặt luật rõ:
|
||||
|
||||
* File có ``schema_version``. Thiếu ⇒ coi là **1** (mọi file đang tồn tại).
|
||||
* Mỗi bước nâng cấp là một hàm ``v1 -> v2``, chạy tuần tự, không nhảy cóc.
|
||||
* **Sao lưu trước khi nâng cấp.** Người dùng lùi về bản app cũ thì bản cũ đọc
|
||||
file mới có thể hỏng — phải còn đường về.
|
||||
* Chỉ nâng, không hạ. File mới hơn app thì báo và dùng nguyên trạng, không cố
|
||||
đoán ngược.
|
||||
|
||||
Bước v1→v2 đầu tiên đi kèm R02-T05: gỡ ``api_key`` khỏi đĩa, đẩy vào
|
||||
``SecretStore``.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import logging
|
||||
import shutil
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable, Dict
|
||||
|
||||
from ..secrets.secret_store import SecretStore, provider_key
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
#: Phiên bản app hiện đang ghi ra.
|
||||
CURRENT_VERSION = 2
|
||||
|
||||
#: Thiếu ``schema_version`` ⇒ file có từ trước khi đánh số.
|
||||
ASSUMED_VERSION = 1
|
||||
|
||||
|
||||
def read_version(data: Dict[str, Any]) -> int:
|
||||
try:
|
||||
return int(data.get("schema_version", ASSUMED_VERSION))
|
||||
except (TypeError, ValueError):
|
||||
return ASSUMED_VERSION
|
||||
|
||||
|
||||
def _v1_to_v2(data: Dict[str, Any], secrets: SecretStore | None) -> Dict[str, Any]:
|
||||
"""Chuyển API key từ file sang kho bí mật — R02-T05.
|
||||
|
||||
Không có kho bí mật thì **không chuyển**: thà để khoá nằm nguyên trong file
|
||||
còn hơn xoá đi rồi người dùng mất khoá mà không hiểu vì sao. File giữ
|
||||
nguyên phiên bản 1, lần chạy sau trên máy có keyring sẽ chuyển.
|
||||
"""
|
||||
if secrets is None or not getattr(secrets, "available", True):
|
||||
log.info("bỏ qua v1→v2: máy này chưa có kho bí mật dùng được")
|
||||
return data
|
||||
|
||||
out = copy.deepcopy(data)
|
||||
moved = []
|
||||
for name, conf in (out.get("providers") or {}).items():
|
||||
if not isinstance(conf, dict):
|
||||
continue
|
||||
key = (conf.get("api_key") or "").strip()
|
||||
# "ollama" là giá trị bù nhìn — Ollama đòi có api_key nhưng bỏ qua nội
|
||||
# dung. Đẩy nó vào keyring chỉ tổ rác.
|
||||
if not key or key == "ollama":
|
||||
continue
|
||||
secrets.set(provider_key(name), key)
|
||||
conf["api_key"] = ""
|
||||
moved.append(name)
|
||||
|
||||
out["schema_version"] = 2
|
||||
if moved:
|
||||
log.info("đã chuyển API key sang kho bí mật: %s", ", ".join(moved))
|
||||
return out
|
||||
|
||||
|
||||
#: {phiên bản nguồn: hàm nâng lên phiên bản kế tiếp}
|
||||
STEPS: Dict[int, Callable[[Dict[str, Any], SecretStore | None], Dict[str, Any]]] = {
|
||||
1: _v1_to_v2,
|
||||
}
|
||||
|
||||
|
||||
def backup(path: Path) -> Path | None:
|
||||
"""Chép file trước khi nâng cấp. Trả về đường dẫn bản sao."""
|
||||
if not path.exists():
|
||||
return None
|
||||
stamp = datetime.now().strftime("%Y%m%d-%H%M%S")
|
||||
target = path.with_suffix(path.suffix + f".v{stamp}.bak")
|
||||
try:
|
||||
shutil.copy2(path, target)
|
||||
return target
|
||||
except OSError as exc:
|
||||
log.warning("không sao lưu được %s: %s", path, exc)
|
||||
return None
|
||||
|
||||
|
||||
def migrate(data: Dict[str, Any], *, secrets: SecretStore | None = None,
|
||||
path: Path | None = None) -> tuple[Dict[str, Any], bool]:
|
||||
"""Nâng ``data`` lên :data:`CURRENT_VERSION`.
|
||||
|
||||
Trả về ``(dữ_liệu, có_đổi_không)``. ``có_đổi_không`` là False thì chỗ gọi
|
||||
khỏi phải ghi lại đĩa.
|
||||
"""
|
||||
version = read_version(data)
|
||||
|
||||
if version > CURRENT_VERSION:
|
||||
# App cũ gặp file mới. Đoán ngược là cách nhanh nhất để mất dữ liệu.
|
||||
log.warning("config phiên bản %s mới hơn app (%s) — dùng nguyên trạng",
|
||||
version, CURRENT_VERSION)
|
||||
return data, False
|
||||
|
||||
if version == CURRENT_VERSION:
|
||||
return data, False
|
||||
|
||||
if path is not None:
|
||||
backup(path)
|
||||
|
||||
changed = False
|
||||
while version < CURRENT_VERSION:
|
||||
step = STEPS.get(version)
|
||||
if step is None:
|
||||
log.warning("thiếu bước nâng cấp từ phiên bản %s — dừng", version)
|
||||
break
|
||||
data = step(data, secrets)
|
||||
new_version = read_version(data)
|
||||
if new_version <= version:
|
||||
# Bước không nâng được phiên bản (ví dụ v1→v2 bỏ qua vì chưa có
|
||||
# keyring). Dừng, đừng lặp vô hạn.
|
||||
break
|
||||
version = new_version
|
||||
changed = True
|
||||
|
||||
return data, changed
|
||||
@@ -0,0 +1,178 @@
|
||||
"""Khung nhìn có kiểu cho từng nhóm cấu hình — R02-T03.
|
||||
|
||||
Vấn đề đang có: khắp nơi viết ``ctx.config.routing.get("switch_mode", "off")``.
|
||||
Gõ sai một chữ thì lặng lẽ nhận giá trị mặc định, không ai biết cho tới khi
|
||||
tính năng "không hiểu sao không chạy". Đếm được **156 lời gọi ``ctx.config.*``
|
||||
trong 29 file** kiểu đó.
|
||||
|
||||
Ở đây mỗi nhóm cấu hình có một lớp: gõ sai tên thuộc tính là lỗi ngay, và kiểu
|
||||
dữ liệu ghi rõ ràng nên đọc code là biết ``confirm_timeout_sec`` là số giây
|
||||
chứ không phải mili giây.
|
||||
|
||||
Cố ý KHÔNG dùng dataclass đông cứng: đây là *khung nhìn* lên dict cấu hình
|
||||
sống, sửa qua đây là sửa vào dict rồi ``save()`` là xuống đĩa. Sao chép thành
|
||||
dataclass thì lại sinh chuyện đồng bộ hai chiều.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict
|
||||
|
||||
|
||||
class _View:
|
||||
"""Khung nhìn lên một nhánh của dict cấu hình."""
|
||||
|
||||
def __init__(self, data: Dict[str, Any]):
|
||||
self._d = data
|
||||
|
||||
def _get(self, key: str, default: Any) -> Any:
|
||||
value = self._d.get(key, default)
|
||||
return default if value is None else value
|
||||
|
||||
def raw(self) -> Dict[str, Any]:
|
||||
"""Dict gốc — dùng khi cần đọc khoá chưa được đưa vào khung nhìn.
|
||||
|
||||
Có mặt để không ai bị kẹt: thiếu thuộc tính thì dùng tạm ``raw()`` rồi
|
||||
mở issue bổ sung, chứ đừng vòng lại ``ctx.config.data``.
|
||||
"""
|
||||
return self._d
|
||||
|
||||
|
||||
class ProviderSettings(_View):
|
||||
"""Một provider: đi đâu, model nào, khoá nào.
|
||||
|
||||
``api_key`` ở đây là thứ ``JsonConfigRepository.provider_conf()`` đã ghép
|
||||
sẵn từ kho bí mật — xem đường A trong ``GammaTeam_decisions.md``.
|
||||
"""
|
||||
|
||||
@property
|
||||
def base_url(self) -> str:
|
||||
return str(self._get("base_url", ""))
|
||||
|
||||
@property
|
||||
def model(self) -> str:
|
||||
return str(self._get("model", ""))
|
||||
|
||||
@property
|
||||
def api_key(self) -> str:
|
||||
return str(self._get("api_key", ""))
|
||||
|
||||
@property
|
||||
def configured(self) -> bool:
|
||||
"""Đủ thông tin để gọi được chưa.
|
||||
|
||||
Ollama chạy cục bộ nên không cần khoá — đó là lý do điều kiện là
|
||||
"có base_url và model", không phải "có api_key".
|
||||
"""
|
||||
return bool(self.base_url and self.model)
|
||||
|
||||
|
||||
class RoutingSettings(_View):
|
||||
"""Định tuyến model tự động (``core/routing/``)."""
|
||||
|
||||
@property
|
||||
def switch_mode(self) -> str:
|
||||
"""``"off"`` | ``"auto"`` | ``"manual"``."""
|
||||
return str(self._get("switch_mode", "off"))
|
||||
|
||||
@switch_mode.setter
|
||||
def switch_mode(self, value: str) -> None:
|
||||
self._d["switch_mode"] = value
|
||||
|
||||
@property
|
||||
def enabled(self) -> bool:
|
||||
return self.switch_mode != "off"
|
||||
|
||||
@property
|
||||
def policy(self) -> str:
|
||||
"""``"balanced"`` | ``"cheap"`` | ``"quality"``…"""
|
||||
return str(self._get("policy", "balanced"))
|
||||
|
||||
@property
|
||||
def min_score_gain(self) -> float:
|
||||
"""Phải hơn model hiện tại bao nhiêu điểm mới đáng đổi."""
|
||||
return float(self._get("min_score_gain", 0.05))
|
||||
|
||||
@property
|
||||
def confirm_timeout_sec(self) -> int:
|
||||
"""GIÂY, không phải mili giây — đọc tên là biết, khỏi phải mò."""
|
||||
return int(self._get("confirm_timeout_sec", 60))
|
||||
|
||||
@property
|
||||
def reassess_interval_hours(self) -> int:
|
||||
return int(self._get("reassess_interval_hours", 24))
|
||||
|
||||
@property
|
||||
def per_provider_concurrency(self) -> int:
|
||||
return int(self._get("per_provider_concurrency", 2))
|
||||
|
||||
@property
|
||||
def judge_provider(self) -> str:
|
||||
return str(self._get("judge_provider", ""))
|
||||
|
||||
@property
|
||||
def judge_model(self) -> str:
|
||||
return str(self._get("judge_model", ""))
|
||||
|
||||
|
||||
class SecuritySettings(_View):
|
||||
"""Chính sách an toàn cho agent (``core/agent_security.py``)."""
|
||||
|
||||
@property
|
||||
def enabled(self) -> bool:
|
||||
return bool(self._get("enabled", True))
|
||||
|
||||
@property
|
||||
def validate_prompt(self) -> bool:
|
||||
return bool(self._get("validate_prompt", True))
|
||||
|
||||
@property
|
||||
def validate_attachments(self) -> bool:
|
||||
return bool(self._get("validate_attachments", True))
|
||||
|
||||
@property
|
||||
def validate_commands(self) -> bool:
|
||||
return bool(self._get("validate_commands", True))
|
||||
|
||||
@property
|
||||
def command_ai_check(self) -> bool:
|
||||
return bool(self._get("command_ai_check", False))
|
||||
|
||||
@property
|
||||
def cowork_confirm_commands(self) -> bool:
|
||||
"""Có hỏi trước khi chạy lệnh không.
|
||||
|
||||
Ứng với ``PolicyOutcome.ASK`` trong
|
||||
``domain/security/tool_policy.py``.
|
||||
"""
|
||||
return bool(self._get("cowork_confirm_commands", True))
|
||||
|
||||
@property
|
||||
def rules_onedrive_url(self) -> str:
|
||||
return str(self._get("rules_onedrive_url", ""))
|
||||
|
||||
@property
|
||||
def admin_email(self) -> str:
|
||||
return str(self._get("admin_email", ""))
|
||||
|
||||
|
||||
class Settings:
|
||||
"""Cửa vào duy nhất cho các nhóm cấu hình có kiểu.
|
||||
|
||||
>>> s = Settings(repo)
|
||||
>>> if s.routing.enabled and s.provider().configured:
|
||||
... ...
|
||||
"""
|
||||
|
||||
def __init__(self, repo):
|
||||
self._repo = repo
|
||||
|
||||
def provider(self, name: str | None = None) -> ProviderSettings:
|
||||
return ProviderSettings(self._repo.provider_conf(name))
|
||||
|
||||
@property
|
||||
def routing(self) -> RoutingSettings:
|
||||
return RoutingSettings(self._repo.routing)
|
||||
|
||||
@property
|
||||
def security(self) -> SecuritySettings:
|
||||
return SecuritySettings(self._repo.agent_security)
|
||||
@@ -0,0 +1,102 @@
|
||||
"""Ghi JSON kiểu không-hỏng-file — R02-T01.
|
||||
|
||||
Vấn đề đang có: ``config.py::save()`` gọi thẳng ``path.write_text(...)``. Hàm
|
||||
đó mở file, cắt cụt về 0 byte, rồi mới ghi nội dung mới. Mất điện, tắt máy, hay
|
||||
process bị kill đúng khoảng giữa thì file cấu hình còn lại **rỗng hoặc ghi dở**
|
||||
— và người dùng mất toàn bộ cấu hình.
|
||||
|
||||
Cách làm ở đây theo đúng thứ tự bắt buộc:
|
||||
|
||||
1. Ghi vào file tạm cùng thư mục (phải cùng ổ đĩa thì bước 3 mới nguyên tử)
|
||||
2. ``flush()`` + ``os.fsync()`` — ép dữ liệu xuống đĩa thật, không nằm trong
|
||||
bộ đệm của hệ điều hành
|
||||
3. ``os.replace()`` — nguyên tử trên cả Windows lẫn POSIX
|
||||
|
||||
Bất kỳ lúc nào chết giữa chừng, file đích vẫn là **bản cũ nguyên vẹn**. Không
|
||||
bao giờ có trạng thái ghi dở.
|
||||
|
||||
Phần đọc có chính sách phục hồi: file hỏng thì giữ lại thành ``.bad`` để còn
|
||||
cứu tay, rồi trả về giá trị mặc định — hỏng cấu hình không được chặn khởi động,
|
||||
đúng như ``config.py`` hiện tại đang làm.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import tempfile
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
class AtomicJsonFile:
|
||||
"""Một file JSON, đọc ghi an toàn.
|
||||
|
||||
>>> f = AtomicJsonFile(Path("cau_hinh.json"))
|
||||
>>> f.write({"theme": "dark"})
|
||||
>>> f.read(default={})
|
||||
{'theme': 'dark'}
|
||||
"""
|
||||
|
||||
def __init__(self, path: Path, *, indent: int = 2):
|
||||
self.path = Path(path)
|
||||
self.indent = indent
|
||||
|
||||
# ---- đọc ------------------------------------------------------------
|
||||
def read(self, default: Any = None) -> Any:
|
||||
"""Nội dung file, hoặc ``default`` nếu chưa có / hỏng.
|
||||
|
||||
Không ném lỗi. File hỏng được đổi tên thành ``<tên>.bad-<thời điểm>``
|
||||
rồi mới trả mặc định — hỏng thì cứu được, chứ đừng ghi đè im lặng.
|
||||
"""
|
||||
if not self.path.exists():
|
||||
return default
|
||||
try:
|
||||
return json.loads(self.path.read_text(encoding="utf-8"))
|
||||
except (json.JSONDecodeError, UnicodeDecodeError):
|
||||
self._quarantine()
|
||||
return default
|
||||
except OSError:
|
||||
# Không đọc được (khoá file, mất quyền) — KHÔNG cách ly, vì file
|
||||
# có thể vẫn tốt nguyên.
|
||||
return default
|
||||
|
||||
def _quarantine(self) -> Path | None:
|
||||
stamp = datetime.now().strftime("%Y%m%d-%H%M%S")
|
||||
target = self.path.with_suffix(self.path.suffix + f".bad-{stamp}")
|
||||
try:
|
||||
os.replace(self.path, target)
|
||||
return target
|
||||
except OSError:
|
||||
return None
|
||||
|
||||
# ---- ghi ------------------------------------------------------------
|
||||
def write(self, data: Any) -> None:
|
||||
"""Ghi ``data``. Hoặc thành công trọn vẹn, hoặc file cũ còn nguyên."""
|
||||
self.path.parent.mkdir(parents=True, exist_ok=True)
|
||||
text = json.dumps(data, indent=self.indent, ensure_ascii=False)
|
||||
|
||||
# File tạm phải nằm CÙNG thư mục: os.replace chỉ nguyên tử trong cùng
|
||||
# một hệ thống tệp. Để ở %TEMP% là có thể rơi sang ổ khác và biến
|
||||
# thành copy + delete — mất luôn tính nguyên tử.
|
||||
fd, tmp_name = tempfile.mkstemp(
|
||||
dir=str(self.path.parent), prefix=f".{self.path.name}.", suffix=".tmp")
|
||||
tmp = Path(tmp_name)
|
||||
try:
|
||||
with os.fdopen(fd, "w", encoding="utf-8") as f:
|
||||
f.write(text)
|
||||
f.flush()
|
||||
os.fsync(f.fileno()) # xuống đĩa thật, không chỉ vào bộ đệm
|
||||
os.replace(tmp, self.path) # nguyên tử
|
||||
except BaseException:
|
||||
# Kể cả KeyboardInterrupt/SystemExit cũng phải dọn file tạm, đừng
|
||||
# để rác .tmp nằm lại cạnh file cấu hình.
|
||||
tmp.unlink(missing_ok=True)
|
||||
raise
|
||||
|
||||
# ---- tiện ích -------------------------------------------------------
|
||||
def exists(self) -> bool:
|
||||
return self.path.exists()
|
||||
|
||||
def __repr__(self) -> str:
|
||||
return f"AtomicJsonFile({self.path})"
|
||||
@@ -1,5 +0,0 @@
|
||||
"""Provider adapters and the central provider catalogue (EPIC R03)."""
|
||||
|
||||
from .provider_registry import ProviderRegistry, default_registry
|
||||
|
||||
__all__ = ["ProviderRegistry", "default_registry"]
|
||||
|
||||
@@ -1,207 +0,0 @@
|
||||
"""ProviderRegistry - the one place a provider is declared (R03-T02).
|
||||
|
||||
Replaces the three-way split between ``providers/factory.py::_REGISTRY``,
|
||||
``config.py::DEFAULT_CONFIG["providers"]`` and ``config.py::PROVIDER_LABELS``
|
||||
with a single catalogue of :class:`ProviderDescriptor` objects plus the
|
||||
implementation class each one maps to.
|
||||
|
||||
Adding a provider is now one entry in :data:`BUILT_IN_PROVIDERS` (declarative
|
||||
facts) and one line in :data:`_IMPLEMENTATIONS` (which class speaks that
|
||||
protocol) - see ``docs/governance/contributor-recipes.md`` (R10-T04).
|
||||
|
||||
Migration note (strangler fig, ADR-001 section 4): this registry does not
|
||||
re-implement any provider. It builds the SAME classes ``providers/factory.py``
|
||||
builds, so both entry points stay behaviourally identical while call sites move
|
||||
over one at a time.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, Iterable, List, Mapping, Optional
|
||||
|
||||
from cowork_local.domain.models.provider_descriptor import (
|
||||
ProviderCapability,
|
||||
ProviderDescriptor,
|
||||
)
|
||||
from cowork_local.providers.base import Provider, ProviderError
|
||||
|
||||
_CAP = ProviderCapability
|
||||
|
||||
# Every provider the app ships with, described once.
|
||||
#
|
||||
# The capability sets are deliberately conservative: a capability listed here is
|
||||
# one the adapter genuinely implements today. Claiming VISION for a provider
|
||||
# whose chat() cannot translate an image block would route an image turn into a
|
||||
# guaranteed failure, so an unimplemented capability must stay off the list.
|
||||
BUILT_IN_PROVIDERS: tuple = (
|
||||
ProviderDescriptor(
|
||||
id="openai_compat",
|
||||
label="OpenAI-compatible (Internal Gateway)",
|
||||
protocol="openai_compat",
|
||||
default_model="gpt-4o-mini",
|
||||
capabilities=frozenset({_CAP.STREAMING, _CAP.TOOLS, _CAP.VISION,
|
||||
_CAP.REASONING, _CAP.MODEL_LISTING}),
|
||||
notes="Any endpoint speaking the OpenAI Chat Completions protocol.",
|
||||
),
|
||||
ProviderDescriptor(
|
||||
id="anthropic",
|
||||
label="Anthropic Claude",
|
||||
protocol="anthropic",
|
||||
default_model="claude-sonnet-4-6",
|
||||
capabilities=frozenset({_CAP.STREAMING, _CAP.TOOLS, _CAP.VISION,
|
||||
_CAP.MODEL_LISTING}),
|
||||
),
|
||||
ProviderDescriptor(
|
||||
id="ollama",
|
||||
label="Ollama (local models)",
|
||||
protocol="openai_compat",
|
||||
default_model="llama3.1",
|
||||
capabilities=frozenset({_CAP.STREAMING, _CAP.TOOLS, _CAP.REASONING,
|
||||
_CAP.MODEL_LISTING}),
|
||||
# Ollama ignores the key, but the OpenAI client layer requires a value,
|
||||
# so the default config ships a placeholder rather than an empty string.
|
||||
requires_api_key=False,
|
||||
local=True,
|
||||
notes="Runs on this machine - no data leaves the device, no token cost.",
|
||||
),
|
||||
ProviderDescriptor(
|
||||
id="github_copilot",
|
||||
label="GitHub Copilot",
|
||||
protocol="openai_compat",
|
||||
default_model="gpt-4o",
|
||||
capabilities=frozenset({_CAP.STREAMING, _CAP.TOOLS, _CAP.MODEL_LISTING}),
|
||||
notes="Paste a Copilot token as the API key.",
|
||||
),
|
||||
ProviderDescriptor(
|
||||
id="codex",
|
||||
label="OpenAI (Codex / GPT)",
|
||||
protocol="openai_compat",
|
||||
default_model="gpt-4o-mini",
|
||||
capabilities=frozenset({_CAP.STREAMING, _CAP.TOOLS, _CAP.VISION,
|
||||
_CAP.REASONING, _CAP.MODEL_LISTING}),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _implementations() -> Dict[str, type]:
|
||||
"""Protocol -> adapter class.
|
||||
|
||||
Imported lazily inside the function because ``providers/anthropic.py`` and
|
||||
``providers/openai_compat.py`` pull in ``requests`` at import time; keeping
|
||||
that out of module import means a test that only inspects descriptors pays
|
||||
no import cost at all.
|
||||
"""
|
||||
from cowork_local.providers.anthropic import AnthropicProvider
|
||||
from cowork_local.providers.openai_compat import OpenAICompatProvider
|
||||
|
||||
return {
|
||||
"openai_compat": OpenAICompatProvider,
|
||||
"anthropic": AnthropicProvider,
|
||||
}
|
||||
|
||||
|
||||
class ProviderRegistry:
|
||||
"""Catalogue of known providers + the factory that instantiates them.
|
||||
|
||||
Intentionally holds no config and no app context: it is a pure lookup table
|
||||
plus a build step, so it can be constructed in a test with a custom
|
||||
descriptor list and no application running.
|
||||
"""
|
||||
|
||||
def __init__(self, descriptors: Optional[Iterable[ProviderDescriptor]] = None) -> None:
|
||||
# Dict preserves declaration order (Python 3.7+), which is the order
|
||||
# Settings lists providers in - so the catalogue order is data, not luck.
|
||||
self._by_id: Dict[str, ProviderDescriptor] = {
|
||||
d.id: d for d in (descriptors if descriptors is not None else BUILT_IN_PROVIDERS)
|
||||
}
|
||||
|
||||
# -- catalogue queries ------------------------------------------------ #
|
||||
def ids(self) -> List[str]:
|
||||
"""Known provider ids, in declaration order."""
|
||||
return list(self._by_id)
|
||||
|
||||
def all(self) -> List[ProviderDescriptor]:
|
||||
"""Every descriptor, in declaration order."""
|
||||
return list(self._by_id.values())
|
||||
|
||||
def get(self, provider_id: str) -> Optional[ProviderDescriptor]:
|
||||
"""The descriptor for ``provider_id``, or None when unknown.
|
||||
|
||||
Returns None rather than raising because the caller is often reacting to
|
||||
a config file that may name a provider from a newer version; the UI
|
||||
should be able to skip it, not crash.
|
||||
"""
|
||||
return self._by_id.get(provider_id)
|
||||
|
||||
def require(self, provider_id: str) -> ProviderDescriptor:
|
||||
"""Like :meth:`get` but raises :class:`ProviderError` when unknown.
|
||||
|
||||
Same error type ``providers/factory.py::build_provider`` already raises,
|
||||
so callers that migrate to the registry keep their existing except clause.
|
||||
"""
|
||||
descriptor = self._by_id.get(provider_id)
|
||||
if descriptor is None:
|
||||
known = ", ".join(self._by_id) or "(none)"
|
||||
raise ProviderError(f"Unsupported provider: {provider_id} (known: {known})")
|
||||
return descriptor
|
||||
|
||||
def labels(self) -> Dict[str, str]:
|
||||
"""``{id: label}`` - the drop-in replacement for ``config.PROVIDER_LABELS``."""
|
||||
return {d.id: d.label for d in self._by_id.values()}
|
||||
|
||||
def supporting(self, capability: ProviderCapability) -> List[ProviderDescriptor]:
|
||||
"""Every descriptor advertising ``capability`` - used to answer "which
|
||||
providers could serve this turn?" before any of them is built."""
|
||||
return [d for d in self._by_id.values() if d.supports(capability)]
|
||||
|
||||
def configured(self, providers_conf: Mapping[str, Mapping[str, Any]]
|
||||
) -> List[ProviderDescriptor]:
|
||||
"""Descriptors whose config section is complete enough to actually call.
|
||||
|
||||
``providers_conf`` is ``AppConfig.data["providers"]``. Passing the raw
|
||||
mapping (not the AppConfig object) keeps this layer independent of the
|
||||
config implementation, which EPIC R02 is rewriting in parallel.
|
||||
"""
|
||||
return [d for d in self._by_id.values()
|
||||
if d.is_configured(providers_conf.get(d.id, {}) or {})]
|
||||
|
||||
# -- construction ----------------------------------------------------- #
|
||||
def build(self, provider_id: str, conf: Mapping[str, Any],
|
||||
model: str = "") -> Provider:
|
||||
"""Instantiate the adapter for ``provider_id``.
|
||||
|
||||
``model`` overrides the configured model for this instance only - that is
|
||||
how the routing layer runs one turn on a different model without mutating
|
||||
the user's saved settings.
|
||||
"""
|
||||
descriptor = self.require(provider_id)
|
||||
impl = _implementations().get(descriptor.protocol)
|
||||
if impl is None: # pragma: no cover - only reachable via a bad descriptor
|
||||
raise ProviderError(
|
||||
f"Provider '{provider_id}' declares unknown protocol "
|
||||
f"'{descriptor.protocol}'."
|
||||
)
|
||||
# Copy before mutating: conf is the caller's live config dict, and
|
||||
# writing the routed model into it would silently change the user's
|
||||
# saved default for every later turn.
|
||||
resolved = dict(conf or {})
|
||||
resolved["model"] = descriptor.resolve_model(conf, model)
|
||||
instance = impl(resolved)
|
||||
# The adapter class is shared by several ids (three of them are
|
||||
# OpenAI-compatible), so its class-level `name` cannot identify which
|
||||
# provider this is. Stamping the instance keeps usage records, audit
|
||||
# entries and routing candidate keys attributed to the right provider.
|
||||
instance.name = descriptor.id
|
||||
return instance
|
||||
|
||||
def describe(self, provider_id: str, conf: Optional[Mapping[str, Any]] = None) -> str:
|
||||
"""One-line description used in logs and error messages."""
|
||||
return self.require(provider_id).describe(conf)
|
||||
|
||||
|
||||
# Shared default instance. Callers that need the built-in catalogue use this
|
||||
# instead of constructing a registry each time; tests build their own with an
|
||||
# explicit descriptor list.
|
||||
default_registry = ProviderRegistry()
|
||||
|
||||
|
||||
__all__ = ["ProviderRegistry", "BUILT_IN_PROVIDERS", "default_registry"]
|
||||
@@ -0,0 +1,86 @@
|
||||
"""SecretStore chạy trên OS Keyring — R02-T04.
|
||||
|
||||
Windows dùng Credential Manager, macOS dùng Keychain, Linux dùng Secret
|
||||
Service. Người dùng cuối không thấy gì khác, nhưng API key thôi nằm trong
|
||||
``config.json`` — đó là điều kiện để qua CASAN Check 1.
|
||||
|
||||
Không phải máy nào cũng có keyring dùng được: Linux chạy headless không có
|
||||
Secret Service, và CI thì gần như chắc chắn không. Nên adapter này **không bao
|
||||
giờ ném lỗi** — không dùng được thì tự báo ``available = False`` và trả về
|
||||
None, để tầng trên hiển thị "chưa lưu được khoá" thay vì sập cả app.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
#: Tên "dịch vụ" trong keyring — mọi khoá của app nằm dưới đây.
|
||||
SERVICE = "cowork-local"
|
||||
|
||||
|
||||
class KeyringAdapter:
|
||||
"""Cài đặt :class:`SecretStore` bằng thư viện ``keyring``.
|
||||
|
||||
>>> store = KeyringAdapter()
|
||||
>>> if store.available:
|
||||
... store.set("provider:openai", "sk-...")
|
||||
"""
|
||||
|
||||
def __init__(self, service: str = SERVICE):
|
||||
self.service = service
|
||||
self._backend = None
|
||||
self._available = False
|
||||
try:
|
||||
import keyring
|
||||
from keyring.backends.fail import Keyring as FailKeyring
|
||||
|
||||
backend = keyring.get_keyring()
|
||||
# backend "fail" là cái keyring trả về khi không tìm được kho nào
|
||||
# dùng được — gọi vào chỉ tổ ném lỗi.
|
||||
if not isinstance(backend, FailKeyring):
|
||||
self._backend = keyring
|
||||
self._available = True
|
||||
else:
|
||||
log.info("keyring không có kho khả dụng trên máy này")
|
||||
except Exception as exc: # noqa: BLE001 — thiếu thư viện, thiếu DBus…
|
||||
log.info("keyring không dùng được: %s", exc)
|
||||
|
||||
@property
|
||||
def available(self) -> bool:
|
||||
"""Có kho bí mật dùng được không.
|
||||
|
||||
Tầng giao diện đọc cờ này để nói cho người dùng biết vì sao ô API key
|
||||
không lưu được, thay vì im lặng làm mất khoá họ vừa nhập.
|
||||
"""
|
||||
return self._available
|
||||
|
||||
# ---- SecretStore ----------------------------------------------------
|
||||
def get(self, key: str) -> str | None:
|
||||
if not self._available:
|
||||
return None
|
||||
try:
|
||||
return self._backend.get_password(self.service, key)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
log.warning("đọc khoá %r thất bại: %s", key, exc)
|
||||
return None
|
||||
|
||||
def set(self, key: str, value: str) -> None:
|
||||
if not self._available:
|
||||
log.warning("không lưu được %r: máy này không có kho bí mật", key)
|
||||
return
|
||||
try:
|
||||
self._backend.set_password(self.service, key, value)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
log.warning("lưu khoá %r thất bại: %s", key, exc)
|
||||
|
||||
def delete(self, key: str) -> None:
|
||||
if not self._available:
|
||||
return
|
||||
try:
|
||||
self._backend.delete_password(self.service, key)
|
||||
except Exception: # noqa: BLE001 — xoá cái không có: bỏ qua
|
||||
pass
|
||||
|
||||
def has(self, key: str) -> bool:
|
||||
return self.get(key) is not None
|
||||
@@ -0,0 +1,46 @@
|
||||
"""Nơi cất credential — interface, chưa phải cài đặt.
|
||||
|
||||
Hợp đồng số 1 của mục chung: chốt hôm nay để N2 và N3 code được ngay, không
|
||||
phải đợi bản Keyring thật (R02-T04, hạn 26/08).
|
||||
|
||||
Vì sao là interface chứ không phải hàm tiện ích: bản thật sẽ gọi OS Keyring —
|
||||
chậm, có thể ném lỗi, và trong test thì không được đụng vào keyring máy thật.
|
||||
Có interface thì test tiêm ``FakeSecretStore`` vào, chạy trong bộ nhớ.
|
||||
|
||||
Quy ước đặt key: ``"provider:<tên>"`` cho API key của provider, ví dụ
|
||||
``"provider:openai"``. Đặt sẵn để không mỗi người tự nghĩ một kiểu.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Protocol, runtime_checkable
|
||||
|
||||
|
||||
def provider_key(name: str) -> str:
|
||||
"""Key chuẩn cho API key của một provider."""
|
||||
return f"provider:{name}"
|
||||
|
||||
|
||||
@runtime_checkable
|
||||
class SecretStore(Protocol):
|
||||
"""Đọc/ghi bí mật. Cài đặt thật: ``KeyringAdapter`` (R02-T04)."""
|
||||
|
||||
def get(self, key: str) -> str | None:
|
||||
"""Giá trị của ``key``, hoặc None nếu chưa có.
|
||||
|
||||
Không được ném lỗi khi thiếu key — thiếu là chuyện bình thường (người
|
||||
dùng chưa nhập API key), không phải sự cố.
|
||||
"""
|
||||
...
|
||||
|
||||
def set(self, key: str, value: str) -> None:
|
||||
"""Lưu ``value``. Ghi đè nếu key đã tồn tại."""
|
||||
...
|
||||
|
||||
def delete(self, key: str) -> None:
|
||||
"""Xoá ``key``. Không có sẵn thì im lặng bỏ qua, không ném lỗi."""
|
||||
...
|
||||
|
||||
def has(self, key: str) -> bool:
|
||||
"""Có key này chưa — dùng cho màn Cài đặt hiển thị trạng thái mà không
|
||||
cần đọc chính giá trị bí mật ra."""
|
||||
...
|
||||
@@ -1,21 +0,0 @@
|
||||
"""Telemetry sinks: where token usage and turn metrics are recorded (EPIC R03)."""
|
||||
|
||||
from .usage_sink import (
|
||||
NullUsageSink,
|
||||
RecordingUsageSink,
|
||||
UsageEvent,
|
||||
UsageEventSink,
|
||||
UsageTrackerSink,
|
||||
default_sink,
|
||||
set_default_sink,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"UsageEvent",
|
||||
"UsageEventSink",
|
||||
"UsageTrackerSink",
|
||||
"NullUsageSink",
|
||||
"RecordingUsageSink",
|
||||
"default_sink",
|
||||
"set_default_sink",
|
||||
]
|
||||
|
||||
@@ -1,229 +0,0 @@
|
||||
"""UsageEventSink - where a turn's token usage goes (R03-T06).
|
||||
|
||||
Today each provider records its own usage inline, in the middle of the streaming
|
||||
loop::
|
||||
|
||||
# providers/openai_compat.py
|
||||
def _record_usage(self, messages, text_parts, tool_acc, usage_seen):
|
||||
from ..core import usage_tracker as ut
|
||||
...
|
||||
ut.record(self.name, self.model, ...)
|
||||
|
||||
Three problems with that shape:
|
||||
|
||||
1. **Hidden side effect.** ``chat()`` looks like a pure request/response call but
|
||||
also writes to the Dashboard's store, so a test of a provider silently
|
||||
appends rows to the developer's real usage history.
|
||||
2. **Duplicated estimation.** The "no usage block from the server, so estimate
|
||||
at ~4 chars/token" fallback is copy-pasted per provider and can drift.
|
||||
3. **One hard-wired destination.** Usage can only ever go to
|
||||
``core.usage_tracker``; a run that wants to bill a workflow, or a test that
|
||||
wants to assert on token counts, has nowhere to plug in.
|
||||
|
||||
This module introduces the seam: providers build a :class:`UsageEvent` and hand
|
||||
it to a :class:`UsageEventSink`. Production wires :class:`UsageTrackerSink`
|
||||
(same destination, same numbers as before); tests wire
|
||||
:class:`RecordingUsageSink` or :class:`NullUsageSink`.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Dict, List, Optional, Protocol, Sequence
|
||||
|
||||
logger = logging.getLogger("cowork_local.telemetry")
|
||||
|
||||
# Rough characters-per-token ratio used when the gateway sends no usage block.
|
||||
# Matches the constant behaviour of ``core.usage_tracker.estimate_tokens`` so
|
||||
# moving the estimation here does not change a single recorded number.
|
||||
_CHARS_PER_TOKEN = 4
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class UsageEvent:
|
||||
"""Token usage for exactly one provider round trip.
|
||||
|
||||
``estimated`` marks a record derived from text length rather than reported by
|
||||
the server. The Dashboard shows the two differently, and conflating them
|
||||
would make cost figures look more precise than they are.
|
||||
"""
|
||||
|
||||
provider: str
|
||||
model: str
|
||||
input_tokens: int = 0
|
||||
output_tokens: int = 0
|
||||
cached_tokens: int = 0
|
||||
estimated: bool = False
|
||||
|
||||
@property
|
||||
def total_tokens(self) -> int:
|
||||
"""Input + output. Cached tokens are a subset of input, not an addition,
|
||||
so adding them here would double-count a cache hit."""
|
||||
return self.input_tokens + self.output_tokens
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
"""JSON-safe projection for logs and for sinks that persist raw events."""
|
||||
return {
|
||||
"provider": self.provider,
|
||||
"model": self.model,
|
||||
"input_tokens": self.input_tokens,
|
||||
"output_tokens": self.output_tokens,
|
||||
"cached_tokens": self.cached_tokens,
|
||||
"estimated": self.estimated,
|
||||
}
|
||||
|
||||
|
||||
class UsageEventSink(Protocol):
|
||||
"""Anything that can absorb a :class:`UsageEvent`.
|
||||
|
||||
Implementations MUST NOT raise: telemetry is observability, and a failure to
|
||||
record usage must never abort the turn that produced it.
|
||||
"""
|
||||
|
||||
def record(self, event: UsageEvent) -> None:
|
||||
"""Absorb one usage event."""
|
||||
|
||||
|
||||
class NullUsageSink:
|
||||
"""Discards everything. The default for tests and headless tooling, so a
|
||||
unit test never writes into the developer's real usage history."""
|
||||
|
||||
def record(self, event: UsageEvent) -> None: # noqa: D102 - see protocol
|
||||
return None
|
||||
|
||||
|
||||
class RecordingUsageSink:
|
||||
"""Keeps events in memory so a test can assert on what was recorded."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.events: List[UsageEvent] = []
|
||||
|
||||
def record(self, event: UsageEvent) -> None: # noqa: D102 - see protocol
|
||||
self.events.append(event)
|
||||
|
||||
@property
|
||||
def total_tokens(self) -> int:
|
||||
"""Sum across every recorded event."""
|
||||
return sum(e.total_tokens for e in self.events)
|
||||
|
||||
|
||||
class UsageTrackerSink:
|
||||
"""Forwards to ``core.usage_tracker`` - the Dashboard's store.
|
||||
|
||||
This is the production sink and the only place that still knows about the
|
||||
legacy tracker module, which is what lets EPIC R10 replace the storage
|
||||
without touching a single provider.
|
||||
"""
|
||||
|
||||
def __init__(self, tracker: Optional[Any] = None) -> None:
|
||||
# Injectable for tests; imported lazily otherwise because the tracker
|
||||
# touches the config directory at import time.
|
||||
self._tracker = tracker
|
||||
|
||||
def _resolve(self) -> Any:
|
||||
if self._tracker is None:
|
||||
from cowork_local.core import usage_tracker
|
||||
|
||||
self._tracker = usage_tracker
|
||||
return self._tracker
|
||||
|
||||
def record(self, event: UsageEvent) -> None:
|
||||
"""Write the event to the usage tracker, swallowing any failure.
|
||||
|
||||
The bare except mirrors the behaviour this replaces (each provider
|
||||
already wrapped its ``ut.record`` call in ``try/except: pass``) but logs
|
||||
at debug level instead of discarding the reason entirely, so a broken
|
||||
Dashboard store can at least be diagnosed.
|
||||
"""
|
||||
try:
|
||||
self._resolve().record(
|
||||
event.provider, event.model,
|
||||
event.input_tokens, event.output_tokens, event.cached_tokens,
|
||||
estimated=event.estimated,
|
||||
)
|
||||
except Exception: # noqa: BLE001 - telemetry must never break a turn
|
||||
logger.debug("usage sink: failed to record %s", event.to_dict(), exc_info=True)
|
||||
|
||||
|
||||
def estimate_tokens(text: str) -> int:
|
||||
"""Approximate token count for ``text`` (~4 characters per token).
|
||||
|
||||
Deliberately identical to ``core.usage_tracker.estimate_tokens`` so that
|
||||
moving estimation into this layer changes no recorded number. Duplicated
|
||||
rather than imported to keep this module free of the legacy dependency;
|
||||
:class:`UsageTrackerSink` is the only bridge back to it.
|
||||
"""
|
||||
return max(0, len(text or "") // _CHARS_PER_TOKEN)
|
||||
|
||||
|
||||
def estimated_event(provider: str, model: str, sent: str, received: str) -> UsageEvent:
|
||||
"""Build an estimated :class:`UsageEvent` from the raw text of a round trip.
|
||||
|
||||
Used when the gateway sends no usage block - most self-hosted OpenAI-compatible
|
||||
servers and Ollama do not.
|
||||
"""
|
||||
return UsageEvent(
|
||||
provider=provider, model=model,
|
||||
input_tokens=estimate_tokens(sent),
|
||||
output_tokens=estimate_tokens(received),
|
||||
cached_tokens=0,
|
||||
estimated=True,
|
||||
)
|
||||
|
||||
|
||||
def openai_usage_event(provider: str, model: str, usage: Dict[str, Any]) -> UsageEvent:
|
||||
"""Build a reported :class:`UsageEvent` from an OpenAI-style usage block."""
|
||||
details = usage.get("prompt_tokens_details") or {}
|
||||
return UsageEvent(
|
||||
provider=provider, model=model,
|
||||
input_tokens=int(usage.get("prompt_tokens", 0) or 0),
|
||||
output_tokens=int(usage.get("completion_tokens", 0) or 0),
|
||||
cached_tokens=int(details.get("cached_tokens", 0) or 0),
|
||||
estimated=False,
|
||||
)
|
||||
|
||||
|
||||
def anthropic_usage_event(provider: str, model: str, usage: Dict[str, Any]) -> UsageEvent:
|
||||
"""Build a reported :class:`UsageEvent` from Anthropic's usage accumulator.
|
||||
|
||||
Anthropic reports input tokens on ``message_start`` and output tokens on
|
||||
``message_delta``, so ``providers/anthropic.py`` accumulates them into a dict
|
||||
keyed ``in``/``out``/``cache`` - this reads that shape.
|
||||
"""
|
||||
return UsageEvent(
|
||||
provider=provider, model=model,
|
||||
input_tokens=int(usage.get("in", 0) or 0),
|
||||
output_tokens=int(usage.get("out", 0) or 0),
|
||||
cached_tokens=int(usage.get("cache", 0) or 0),
|
||||
estimated=False,
|
||||
)
|
||||
|
||||
|
||||
# The sink providers use unless one is injected. A module-level default keeps
|
||||
# the change to the provider classes to a single attribute, and lets a test swap
|
||||
# the destination process-wide with one monkeypatch.
|
||||
default_sink: UsageEventSink = UsageTrackerSink()
|
||||
|
||||
|
||||
def set_default_sink(sink: UsageEventSink) -> UsageEventSink:
|
||||
"""Replace the process-wide default sink; returns the previous one so a
|
||||
caller (or fixture) can restore it."""
|
||||
global default_sink
|
||||
previous = default_sink
|
||||
default_sink = sink
|
||||
return previous
|
||||
|
||||
|
||||
__all__ = [
|
||||
"UsageEvent",
|
||||
"UsageEventSink",
|
||||
"UsageTrackerSink",
|
||||
"NullUsageSink",
|
||||
"RecordingUsageSink",
|
||||
"estimate_tokens",
|
||||
"estimated_event",
|
||||
"openai_usage_event",
|
||||
"anthropic_usage_event",
|
||||
"default_sink",
|
||||
"set_default_sink",
|
||||
]
|
||||
@@ -0,0 +1,5 @@
|
||||
"""Provider-neutral Project Context MCP server template."""
|
||||
|
||||
from .server import build_server, dispatch
|
||||
|
||||
__all__ = ["build_server", "dispatch"]
|
||||
@@ -0,0 +1,106 @@
|
||||
"""Shared, stable boundary used by all Project Context tool work packages."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Callable
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
from typing import Any, Protocol
|
||||
|
||||
from pydantic import AnyUrl, BaseModel, ConfigDict, Field
|
||||
|
||||
|
||||
class ContractModel(BaseModel):
|
||||
"""Strict immutable model so provider-specific fields cannot leak to the Agent."""
|
||||
|
||||
model_config = ConfigDict(extra="forbid", frozen=True)
|
||||
|
||||
|
||||
class IdentityContext(ContractModel):
|
||||
actor_id: str = Field(min_length=1, max_length=256)
|
||||
org_unit: str = Field(min_length=1, max_length=128)
|
||||
customer: str = Field(min_length=1, max_length=128)
|
||||
project: str = Field(min_length=1, max_length=128)
|
||||
granted_scopes: frozenset[str]
|
||||
|
||||
|
||||
class SourceCitation(ContractModel):
|
||||
system: str = Field(min_length=1, max_length=64)
|
||||
url: AnyUrl
|
||||
revision: str = Field(min_length=1, max_length=256)
|
||||
retrieved_at: datetime
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class DispatchResult:
|
||||
ok: bool
|
||||
payload: dict[str, Any]
|
||||
|
||||
|
||||
class PolicyDecisionPoint(Protocol):
|
||||
def decide(self, identity: IdentityContext, tool_name: str, project_id: str) -> bool: ...
|
||||
|
||||
|
||||
class CredentialResolver(Protocol):
|
||||
def resolve(self, identity: IdentityContext, tool_name: str) -> Any: ...
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ProjectContextRuntime:
|
||||
identity: IdentityContext
|
||||
policy: PolicyDecisionPoint
|
||||
credential_resolver: CredentialResolver
|
||||
|
||||
|
||||
class ProviderError(RuntimeError):
|
||||
"""A provider failure with a caller-safe message and retry classification."""
|
||||
|
||||
def __init__(self, code: str, message: str, *, retryable: bool) -> None:
|
||||
super().__init__(message)
|
||||
self.code = code
|
||||
self.safe_message = message
|
||||
self.retryable = retryable
|
||||
|
||||
|
||||
ToolHandler = Callable[[ContractModel, Any], dict[str, Any]]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ToolTemplate:
|
||||
name: str
|
||||
description: str
|
||||
input_model: type[ContractModel]
|
||||
output_model: type[ContractModel]
|
||||
handler: ToolHandler
|
||||
|
||||
def declaration(self) -> dict[str, Any]:
|
||||
return {
|
||||
"name": self.name,
|
||||
"description": self.description,
|
||||
"inputSchema": self.input_model.model_json_schema(),
|
||||
"outputSchema": self.output_model.model_json_schema(),
|
||||
}
|
||||
|
||||
|
||||
def error_result(
|
||||
code: str,
|
||||
*,
|
||||
category: str,
|
||||
retryable: bool,
|
||||
message: str,
|
||||
suggested_action: str,
|
||||
correlation_id: str,
|
||||
) -> DispatchResult:
|
||||
return DispatchResult(
|
||||
ok=False,
|
||||
payload={
|
||||
"error": {
|
||||
"code": code,
|
||||
"category": category,
|
||||
"retryable": retryable,
|
||||
"message": message,
|
||||
"suggested_action": suggested_action,
|
||||
"correlation_id": correlation_id,
|
||||
}
|
||||
},
|
||||
)
|
||||
@@ -0,0 +1 @@
|
||||
"""One provider module per member-owned tool work package."""
|
||||
@@ -0,0 +1,25 @@
|
||||
"""Provider boundary owned with get_project_change_context."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Protocol
|
||||
|
||||
from ..foundation import IdentityContext, ProviderError
|
||||
|
||||
|
||||
class ChangeProvider(Protocol):
|
||||
def get_change_context(self, **arguments: Any) -> dict[str, Any]: ...
|
||||
|
||||
|
||||
class UnconfiguredChangeProvider:
|
||||
def get_change_context(self, **arguments: Any) -> dict[str, Any]:
|
||||
raise ProviderError(
|
||||
"UNAVAILABLE",
|
||||
"The change provider is not configured for this environment.",
|
||||
retryable=False,
|
||||
)
|
||||
|
||||
|
||||
def build_provider(identity: IdentityContext) -> ChangeProvider:
|
||||
"""Replace only this factory when wiring the approved read-only Git adapter."""
|
||||
return UnconfiguredChangeProvider()
|
||||
@@ -0,0 +1,25 @@
|
||||
"""Provider boundary owned with get_project_issue_context."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Protocol
|
||||
|
||||
from ..foundation import IdentityContext, ProviderError
|
||||
|
||||
|
||||
class IssueProvider(Protocol):
|
||||
def get_issue_context(self, **arguments: Any) -> dict[str, Any]: ...
|
||||
|
||||
|
||||
class UnconfiguredIssueProvider:
|
||||
def get_issue_context(self, **arguments: Any) -> dict[str, Any]:
|
||||
raise ProviderError(
|
||||
"UNAVAILABLE",
|
||||
"The issue provider is not configured for this environment.",
|
||||
retryable=False,
|
||||
)
|
||||
|
||||
|
||||
def build_provider(identity: IdentityContext) -> IssueProvider:
|
||||
"""Replace only this factory when wiring the approved read-only issue adapter."""
|
||||
return UnconfiguredIssueProvider()
|
||||
@@ -0,0 +1,25 @@
|
||||
"""Provider boundary owned with search_project_knowledge."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Protocol
|
||||
|
||||
from ..foundation import IdentityContext, ProviderError
|
||||
|
||||
|
||||
class KnowledgeProvider(Protocol):
|
||||
def search_knowledge(self, **arguments: Any) -> dict[str, Any]: ...
|
||||
|
||||
|
||||
class UnconfiguredKnowledgeProvider:
|
||||
def search_knowledge(self, **arguments: Any) -> dict[str, Any]:
|
||||
raise ProviderError(
|
||||
"UNAVAILABLE",
|
||||
"The knowledge provider is not configured for this environment.",
|
||||
retryable=False,
|
||||
)
|
||||
|
||||
|
||||
def build_provider(identity: IdentityContext) -> KnowledgeProvider:
|
||||
"""Replace only this factory when wiring approved project retrieval."""
|
||||
return UnconfiguredKnowledgeProvider()
|
||||
@@ -0,0 +1,23 @@
|
||||
"""Immutable registry composed before member work starts to prevent merge conflicts."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from types import MappingProxyType
|
||||
from typing import Any
|
||||
|
||||
from .foundation import ToolTemplate
|
||||
from .tools.change_context import TOOL as CHANGE_CONTEXT_TOOL
|
||||
from .tools.issue_context import TOOL as ISSUE_CONTEXT_TOOL
|
||||
from .tools.knowledge_search import TOOL as KNOWLEDGE_SEARCH_TOOL
|
||||
|
||||
TOOLS: tuple[ToolTemplate, ...] = (
|
||||
ISSUE_CONTEXT_TOOL,
|
||||
KNOWLEDGE_SEARCH_TOOL,
|
||||
CHANGE_CONTEXT_TOOL,
|
||||
)
|
||||
TOOLS_BY_NAME = MappingProxyType({tool.name: tool for tool in TOOLS})
|
||||
TOOL_NAMES = tuple(tool.name for tool in TOOLS)
|
||||
|
||||
|
||||
def tool_declarations() -> list[dict[str, Any]]:
|
||||
return [tool.declaration() for tool in TOOLS]
|
||||
@@ -0,0 +1,74 @@
|
||||
"""Fail-closed identity, policy, and provider resolution for the template server."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
from collections.abc import Callable
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
from .foundation import IdentityContext, ProjectContextRuntime, ProviderError
|
||||
from .providers.change import build_provider as build_change_provider
|
||||
from .providers.issue import build_provider as build_issue_provider
|
||||
from .providers.knowledge import build_provider as build_knowledge_provider
|
||||
|
||||
MINIMUM_PYTHON = (3, 11)
|
||||
|
||||
|
||||
def require_supported_python(version_info: tuple[int, ...] | None = None) -> None:
|
||||
"""Fail with an actionable message before the MCP server starts."""
|
||||
current = version_info or tuple(sys.version_info[:3])
|
||||
if current[:2] < MINIMUM_PYTHON:
|
||||
raise RuntimeError(
|
||||
"Project Context MCP requires Python 3.11 or newer; "
|
||||
f"current runtime is {current[0]}.{current[1]}"
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ProjectScopePolicy:
|
||||
"""Pilot policy: read scope and exact identity-bound project are both mandatory."""
|
||||
|
||||
def decide(self, identity: IdentityContext, tool_name: str, project_id: str) -> bool:
|
||||
return "read" in identity.granted_scopes and project_id == identity.project
|
||||
|
||||
|
||||
PROVIDER_FACTORIES: dict[str, Callable[[IdentityContext], Any]] = {
|
||||
"get_project_issue_context": build_issue_provider,
|
||||
"search_project_knowledge": build_knowledge_provider,
|
||||
"get_project_change_context": build_change_provider,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ProjectProviderResolver:
|
||||
def resolve(self, identity: IdentityContext, tool_name: str) -> Any:
|
||||
factory = PROVIDER_FACTORIES.get(tool_name)
|
||||
if factory is None:
|
||||
raise ProviderError("NOT_FOUND", "The requested tool is not registered.", retryable=False)
|
||||
return factory(identity)
|
||||
|
||||
|
||||
def _required_environment(name: str) -> str:
|
||||
value = os.environ.get(name, "").strip()
|
||||
if not value:
|
||||
raise RuntimeError(f"Project Context MCP cannot start: required setting {name} is missing")
|
||||
return value
|
||||
|
||||
|
||||
def default_runtime() -> ProjectContextRuntime:
|
||||
"""Build immutable runtime state; missing identity configuration fails at boot."""
|
||||
require_supported_python()
|
||||
identity = IdentityContext(
|
||||
actor_id=_required_environment("COWORK_MCP_ACTOR_ID"),
|
||||
org_unit=_required_environment("COWORK_MCP_ORG_UNIT"),
|
||||
customer=_required_environment("COWORK_MCP_CUSTOMER"),
|
||||
project=_required_environment("COWORK_MCP_PROJECT"),
|
||||
granted_scopes=frozenset({"read"}),
|
||||
)
|
||||
return ProjectContextRuntime(
|
||||
identity=identity,
|
||||
policy=ProjectScopePolicy(),
|
||||
credential_resolver=ProjectProviderResolver(),
|
||||
)
|
||||
@@ -0,0 +1,142 @@
|
||||
"""Low-level MCP stdio adapter around the transport-agnostic Project Context core."""
|
||||
|
||||
# ruff: noqa: UP045 -- Optional keeps the template importable with Pydantic on Python 3.9.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any, Optional
|
||||
from uuid import uuid4
|
||||
|
||||
from pydantic import ValidationError
|
||||
|
||||
from .foundation import (
|
||||
DispatchResult,
|
||||
ProjectContextRuntime,
|
||||
ProviderError,
|
||||
error_result,
|
||||
)
|
||||
from .registry import TOOLS_BY_NAME, tool_declarations
|
||||
from .runtime import default_runtime, require_supported_python
|
||||
|
||||
|
||||
def dispatch(
|
||||
name: str,
|
||||
arguments: dict[str, Any],
|
||||
runtime: ProjectContextRuntime,
|
||||
) -> DispatchResult:
|
||||
"""Validate → authorize → resolve provider → execute → validate output."""
|
||||
correlation_id = str(uuid4())
|
||||
tool = TOOLS_BY_NAME.get(name)
|
||||
if tool is None:
|
||||
return error_result(
|
||||
"NOT_FOUND",
|
||||
category="NOT_FOUND",
|
||||
retryable=False,
|
||||
message="The requested MCP tool is not registered.",
|
||||
suggested_action="Refresh the tool list and choose one of the advertised tools.",
|
||||
correlation_id=correlation_id,
|
||||
)
|
||||
|
||||
try:
|
||||
validated_input = tool.input_model.model_validate(arguments or {})
|
||||
except ValidationError:
|
||||
return error_result(
|
||||
"INVALID_INPUT",
|
||||
category="INVALID_INPUT",
|
||||
retryable=False,
|
||||
message="The tool arguments do not match the published input contract.",
|
||||
suggested_action="Correct the required fields and value bounds, then call again.",
|
||||
correlation_id=correlation_id,
|
||||
)
|
||||
|
||||
project_id = str(validated_input.project_id)
|
||||
if not runtime.policy.decide(runtime.identity, name, project_id):
|
||||
return error_result(
|
||||
"DENIED",
|
||||
category="DENIED",
|
||||
retryable=False,
|
||||
message="The project is outside the caller's approved scope.",
|
||||
suggested_action="Use an approved project or ask the project owner for access.",
|
||||
correlation_id=correlation_id,
|
||||
)
|
||||
|
||||
try:
|
||||
provider = runtime.credential_resolver.resolve(runtime.identity, name)
|
||||
raw_output = tool.handler(validated_input, provider)
|
||||
except ProviderError as exc:
|
||||
return error_result(
|
||||
exc.code,
|
||||
category=exc.code,
|
||||
retryable=exc.retryable,
|
||||
message=exc.safe_message,
|
||||
suggested_action="Check the approved provider configuration and retry if allowed.",
|
||||
correlation_id=correlation_id,
|
||||
)
|
||||
except Exception: # noqa: BLE001 - provider failures must not crash or leak into the agent turn
|
||||
return error_result(
|
||||
"UPSTREAM_ERROR",
|
||||
category="UPSTREAM_ERROR",
|
||||
retryable=False,
|
||||
message="The approved provider could not complete the request.",
|
||||
suggested_action="Check the correlation ID in server logs; do not resend credentials.",
|
||||
correlation_id=correlation_id,
|
||||
)
|
||||
|
||||
try:
|
||||
output_with_trace = {**raw_output, "correlation_id": correlation_id}
|
||||
validated_output = tool.output_model.model_validate(output_with_trace)
|
||||
except ValidationError:
|
||||
return error_result(
|
||||
"UPSTREAM_ERROR",
|
||||
category="UPSTREAM_ERROR",
|
||||
retryable=False,
|
||||
message="The provider response did not match the published output contract.",
|
||||
suggested_action="Fix the provider mapping before retrying the request.",
|
||||
correlation_id=correlation_id,
|
||||
)
|
||||
return DispatchResult(ok=True, payload=validated_output.model_dump(mode="json"))
|
||||
|
||||
|
||||
def build_server(runtime: Optional[ProjectContextRuntime] = None):
|
||||
from mcp import types
|
||||
from mcp.server.lowlevel import Server
|
||||
|
||||
require_supported_python()
|
||||
app_runtime = runtime or default_runtime()
|
||||
app = Server("project_context")
|
||||
|
||||
@app.list_tools()
|
||||
async def list_tools() -> list[types.Tool]:
|
||||
return [types.Tool(**declaration) for declaration in tool_declarations()]
|
||||
|
||||
@app.call_tool()
|
||||
async def call_tool(name: str, arguments: dict[str, Any]) -> types.CallToolResult:
|
||||
result = dispatch(name, arguments or {}, app_runtime)
|
||||
return types.CallToolResult(
|
||||
content=[types.TextContent(
|
||||
type="text",
|
||||
text=json.dumps(result.payload, ensure_ascii=False, separators=(",", ":")),
|
||||
)],
|
||||
structuredContent=result.payload if result.ok else None,
|
||||
isError=not result.ok,
|
||||
)
|
||||
|
||||
return app
|
||||
|
||||
|
||||
def main() -> None:
|
||||
import anyio
|
||||
from mcp.server.stdio import stdio_server
|
||||
|
||||
app = build_server()
|
||||
|
||||
async def _run() -> None:
|
||||
async with stdio_server() as (read, write):
|
||||
await app.run(read, write, app.create_initialization_options())
|
||||
|
||||
anyio.run(_run)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1 @@
|
||||
"""Independent tool modules; ownership is documented in the team guide."""
|
||||
@@ -0,0 +1,55 @@
|
||||
"""Member C work package: get_project_change_context."""
|
||||
|
||||
# ruff: noqa: UP045 -- Optional keeps Pydantic model evaluation compatible with Python 3.9.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Literal, Optional
|
||||
|
||||
from pydantic import Field
|
||||
|
||||
from ..foundation import ContractModel, SourceCitation, ToolTemplate
|
||||
|
||||
|
||||
class ChangeContextInput(ContractModel):
|
||||
project_id: str = Field(min_length=1, max_length=128)
|
||||
change_id: str = Field(min_length=1, max_length=128)
|
||||
detail: Literal["summary", "standard", "full"] = "standard"
|
||||
cursor: Optional[str] = Field(default=None, max_length=2048)
|
||||
|
||||
|
||||
class ChangeContextOutput(ContractModel):
|
||||
correlation_id: str
|
||||
project_id: str
|
||||
change_id: str
|
||||
change_type: Literal["commit", "pull-request", "merge-request"]
|
||||
title: str
|
||||
state: str
|
||||
summary: str
|
||||
authors: tuple[str, ...]
|
||||
files: tuple[str, ...]
|
||||
commits: tuple[str, ...]
|
||||
related_issues: tuple[str, ...]
|
||||
source: SourceCitation
|
||||
truncated: bool
|
||||
returned: int = Field(ge=0)
|
||||
remaining: int = Field(ge=0)
|
||||
next_cursor: Optional[str] = None
|
||||
|
||||
|
||||
def _handle(arguments: ContractModel, provider: Any) -> dict[str, Any]:
|
||||
request = ChangeContextInput.model_validate(arguments)
|
||||
return provider.get_change_context(**request.model_dump())
|
||||
|
||||
|
||||
TOOL = ToolTemplate(
|
||||
name="get_project_change_context",
|
||||
description=(
|
||||
"Returns provider-neutral context for one authorized commit, pull request, or merge request "
|
||||
"with changed files, commits, related issues, and a pinned source. Use when an exact change "
|
||||
"identifier is known. Do not use for issue details or free-text document search."
|
||||
),
|
||||
input_model=ChangeContextInput,
|
||||
output_model=ChangeContextOutput,
|
||||
handler=_handle,
|
||||
)
|
||||
@@ -0,0 +1,59 @@
|
||||
"""Member A work package: get_project_issue_context."""
|
||||
|
||||
# ruff: noqa: UP045 -- Optional keeps Pydantic model evaluation compatible with Python 3.9.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Literal, Optional
|
||||
|
||||
from pydantic import Field
|
||||
|
||||
from ..foundation import ContractModel, SourceCitation, ToolTemplate
|
||||
|
||||
|
||||
class IssueContextInput(ContractModel):
|
||||
project_id: str = Field(min_length=1, max_length=128)
|
||||
issue_key: str = Field(min_length=1, max_length=128)
|
||||
detail: Literal["summary", "standard", "full"] = "standard"
|
||||
cursor: Optional[str] = Field(default=None, max_length=2048)
|
||||
|
||||
|
||||
class RelatedItem(ContractModel):
|
||||
item_id: str
|
||||
relation: str
|
||||
title: str
|
||||
url: str
|
||||
|
||||
|
||||
class IssueContextOutput(ContractModel):
|
||||
correlation_id: str
|
||||
project_id: str
|
||||
issue_key: str
|
||||
title: str
|
||||
status: str
|
||||
description: str
|
||||
acceptance_criteria: tuple[str, ...]
|
||||
related: tuple[RelatedItem, ...]
|
||||
source: SourceCitation
|
||||
truncated: bool
|
||||
returned: int = Field(ge=0)
|
||||
remaining: int = Field(ge=0)
|
||||
next_cursor: Optional[str] = None
|
||||
|
||||
|
||||
def _handle(arguments: ContractModel, provider: Any) -> dict[str, Any]:
|
||||
request = IssueContextInput.model_validate(arguments)
|
||||
return provider.get_issue_context(**request.model_dump())
|
||||
|
||||
|
||||
TOOL = ToolTemplate(
|
||||
name="get_project_issue_context",
|
||||
description=(
|
||||
"Returns one authorized work item's title, state, description, acceptance criteria, "
|
||||
"related items, and pinned source. Use when an exact issue key is known. Do not use for "
|
||||
"free-text knowledge search or Git change review."
|
||||
),
|
||||
input_model=IssueContextInput,
|
||||
output_model=IssueContextOutput,
|
||||
handler=_handle,
|
||||
)
|
||||
@@ -0,0 +1,58 @@
|
||||
"""Member B work package: search_project_knowledge."""
|
||||
|
||||
# ruff: noqa: UP045 -- Optional keeps Pydantic model evaluation compatible with Python 3.9.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Literal, Optional
|
||||
|
||||
from pydantic import Field
|
||||
|
||||
from ..foundation import ContractModel, SourceCitation, ToolTemplate
|
||||
|
||||
|
||||
class KnowledgeSearchInput(ContractModel):
|
||||
project_id: str = Field(min_length=1, max_length=128)
|
||||
query: str = Field(min_length=2, max_length=1000)
|
||||
detail: Literal["summary", "standard", "full"] = "standard"
|
||||
top_k: int = Field(default=5, ge=1, le=20)
|
||||
language: Optional[Literal["en", "ja", "vi"]] = None
|
||||
cursor: Optional[str] = Field(default=None, max_length=2048)
|
||||
|
||||
|
||||
class KnowledgeItem(ContractModel):
|
||||
document_id: str
|
||||
chunk_id: str
|
||||
title: str
|
||||
excerpt: str
|
||||
score: float = Field(ge=0, le=1)
|
||||
source: SourceCitation
|
||||
|
||||
|
||||
class KnowledgeSearchOutput(ContractModel):
|
||||
correlation_id: str
|
||||
project_id: str
|
||||
query: str
|
||||
items: tuple[KnowledgeItem, ...]
|
||||
truncated: bool
|
||||
returned: int = Field(ge=0)
|
||||
remaining: int = Field(ge=0)
|
||||
next_cursor: Optional[str] = None
|
||||
|
||||
|
||||
def _handle(arguments: ContractModel, provider: Any) -> dict[str, Any]:
|
||||
request = KnowledgeSearchInput.model_validate(arguments)
|
||||
return provider.search_knowledge(**request.model_dump())
|
||||
|
||||
|
||||
TOOL = ToolTemplate(
|
||||
name="search_project_knowledge",
|
||||
description=(
|
||||
"Searches approved knowledge for one authorized project and returns ranked excerpts with "
|
||||
"pinned citations. Use for requirements, design notes, or runbooks when no exact issue is "
|
||||
"known. Do not use for issue details or Git change review."
|
||||
),
|
||||
input_model=KnowledgeSearchInput,
|
||||
output_model=KnowledgeSearchOutput,
|
||||
handler=_handle,
|
||||
)
|
||||
@@ -0,0 +1,9 @@
|
||||
"""Stable module entry point for ``python -m cowork_local.mcp_servers.project_context_server``."""
|
||||
|
||||
from .project_context.server import build_server, dispatch, main
|
||||
|
||||
__all__ = ["build_server", "dispatch", "main"]
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1 @@
|
||||
"""presentation/ — Widget Qt. Chỉ gọi xuống application, không gọi thẳng infrastructure."""
|
||||
+14
-12
@@ -292,19 +292,21 @@ class AnthropicProvider(Provider):
|
||||
args = {"_raw": b["json"]}
|
||||
tool_calls.append({"id": b["id"], "name": b["name"], "arguments": args})
|
||||
|
||||
# Dashboard usage event — real counts from the stream's usage events
|
||||
# (input arrives on message_start, output on message_delta), else a
|
||||
# ~4 chars/token estimate. Delivery is the sink's job (R03-T06), so this
|
||||
# only translates Anthropic's wire shape into a canonical UsageEvent.
|
||||
from ..infrastructure.telemetry import usage_sink as telemetry
|
||||
# Dashboard usage event — real counts from the stream's usage events,
|
||||
# else a ~4 chars/token estimate. Never breaks the turn.
|
||||
try:
|
||||
from ..core import usage_tracker as ut
|
||||
|
||||
if usage_seen:
|
||||
event = telemetry.anthropic_usage_event(self.name, self.model, usage_seen)
|
||||
else:
|
||||
sent = json.dumps(payload.get("messages", []), ensure_ascii=False)
|
||||
got = "".join(text_parts) + "".join(b["json"] for b in blocks.values())
|
||||
event = telemetry.estimated_event(self.name, self.model, sent, got)
|
||||
self._emit_usage(event)
|
||||
if usage_seen:
|
||||
ut.record(self.name, self.model, usage_seen.get("in", 0),
|
||||
usage_seen.get("out", 0), usage_seen.get("cache", 0))
|
||||
else:
|
||||
sent = json.dumps(payload.get("messages", []), ensure_ascii=False)
|
||||
got = "".join(text_parts) + "".join(b["json"] for b in blocks.values())
|
||||
ut.record(self.name, self.model, ut.estimate_tokens(sent),
|
||||
ut.estimate_tokens(got), 0, estimated=True)
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
|
||||
return {"role": "assistant", "content": "".join(text_parts), "tool_calls": tool_calls}
|
||||
|
||||
|
||||
@@ -224,12 +224,6 @@ class Provider:
|
||||
# silently swallowing the error — Settings' "Test connection" / "Load
|
||||
# models" surfaces this so "model won't load" has a concrete reason.
|
||||
self.last_error = ""
|
||||
# Where this provider's token usage goes (R03-T06). None means "the
|
||||
# process-wide default sink", resolved lazily in _emit_usage so that a
|
||||
# test can swap the destination without rebuilding every provider.
|
||||
# Set it per instance to bill one run somewhere else (a workflow, a
|
||||
# scheduled task) without touching global state.
|
||||
self.usage_sink = None
|
||||
|
||||
def chat(
|
||||
self,
|
||||
@@ -280,24 +274,6 @@ class Provider:
|
||||
return True, f"OK — {len(models)} model(s) available."
|
||||
return False, "No models returned. Check base_url/API key and network access."
|
||||
|
||||
# -- telemetry -----------------------------------------------------
|
||||
def _emit_usage(self, event) -> None:
|
||||
"""Hand one ``UsageEvent`` to this provider's usage sink.
|
||||
|
||||
Never raises: recording how many tokens a turn cost must not be able to
|
||||
fail the turn itself. Falls back to the process-wide default sink so
|
||||
existing call sites keep reporting to the Dashboard exactly as before
|
||||
(see infrastructure/telemetry/usage_sink.py)."""
|
||||
try:
|
||||
sink = self.usage_sink
|
||||
if sink is None:
|
||||
from ..infrastructure.telemetry import usage_sink as telemetry
|
||||
|
||||
sink = telemetry.default_sink
|
||||
sink.record(event)
|
||||
except Exception: # noqa: BLE001 — telemetry is never worth a failed turn
|
||||
pass
|
||||
|
||||
# -- shared helpers ------------------------------------------------
|
||||
@staticmethod
|
||||
def _is_cancelled(cancel) -> bool:
|
||||
|
||||
+15
-18
@@ -268,25 +268,22 @@ class OpenAICompatProvider(Provider):
|
||||
def _record_usage(self, messages, text_parts, tool_acc, usage_seen) -> None:
|
||||
"""One Dashboard usage event per turn: real counts when the server's
|
||||
final chunk carried a "usage" block, a ~4 chars/token estimate
|
||||
otherwise.
|
||||
otherwise. Never breaks the turn."""
|
||||
try:
|
||||
from ..core import usage_tracker as ut
|
||||
|
||||
Building the event and delivering it are now separate concerns (R03-T06):
|
||||
this method only translates THIS provider's wire shape into a canonical
|
||||
``UsageEvent``; where it ends up is the sink's decision, so a test can
|
||||
assert on token counts without writing to the real Dashboard store."""
|
||||
from ..infrastructure.telemetry import usage_sink as telemetry
|
||||
|
||||
if usage_seen:
|
||||
event = telemetry.openai_usage_event(self.name, self.model, usage_seen)
|
||||
else:
|
||||
# No usage block from the gateway (self-hosted servers and Ollama
|
||||
# never send one) - fall back to estimating from the raw text of
|
||||
# both directions, tool-call arguments included since the model was
|
||||
# billed for generating them.
|
||||
sent = json.dumps(self._to_api_messages(messages), ensure_ascii=False)
|
||||
got = "".join(text_parts) + "".join(s["args"] for s in tool_acc.values())
|
||||
event = telemetry.estimated_event(self.name, self.model, sent, got)
|
||||
self._emit_usage(event)
|
||||
if usage_seen:
|
||||
ut.record(self.name, self.model,
|
||||
usage_seen.get("prompt_tokens", 0),
|
||||
usage_seen.get("completion_tokens", 0),
|
||||
(usage_seen.get("prompt_tokens_details") or {}).get("cached_tokens", 0))
|
||||
else:
|
||||
sent = json.dumps(self._to_api_messages(messages), ensure_ascii=False)
|
||||
got = "".join(text_parts) + "".join(s["args"] for s in tool_acc.values())
|
||||
ut.record(self.name, self.model, ut.estimate_tokens(sent),
|
||||
ut.estimate_tokens(got), 0, estimated=True)
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
|
||||
def list_models(self):
|
||||
self.last_error = ""
|
||||
|
||||
@@ -0,0 +1,204 @@
|
||||
"""CASAN Check 1 — không được có credential nào nằm phơi trong repo.
|
||||
|
||||
Team Gamma chủ trì check này (hạn: 30/08). Viết sẵn từ 21/08 để chạy được liên
|
||||
tục trong lúc chuyển API key sang Keyring (R02-T05), thay vì tới ngày cổng mới
|
||||
chạy lần đầu rồi mới biết còn sót.
|
||||
|
||||
Quét gì:
|
||||
* file cấu hình đã commit: ``*.json`` ``*.jsonl`` ``*.yaml`` ``*.yml`` ``*.env``
|
||||
* mã nguồn Python — chỗ gán chuỗi cho biến tên như api_key / token / secret
|
||||
|
||||
Tìm hai loại:
|
||||
1. Chuỗi có hình dạng credential thật (sk-…, ghp_…, xoxb-…, AKIA…, JWT…)
|
||||
2. Trường tên nhạy cảm mà giá trị không rỗng và không phải placeholder
|
||||
|
||||
Bỏ qua: chuỗi rỗng, placeholder ("your-key-here", "changeme"…), giá trị hằng
|
||||
không phải bí mật (Ollama đòi có api_key nhưng bỏ qua nội dung).
|
||||
|
||||
Chạy: python scripts/audit_security.py [--json]
|
||||
Mã thoát: 0 = sạch, 1 = có phát hiện.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
|
||||
# console Windows hay là cp932/cp1258; ép UTF-8 để không chết giữa báo cáo
|
||||
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
||||
from pathlib import Path
|
||||
|
||||
REPO = Path(__file__).resolve().parent.parent
|
||||
|
||||
SKIP_DIRS = {".git", "__pycache__", "node_modules", ".venv", "venv", "build",
|
||||
"dist", ".pytest_cache", ".mypy_cache", "cowork-local-gitea"}
|
||||
CONFIG_SUFFIX = {".json", ".jsonl", ".yaml", ".yml", ".env"}
|
||||
|
||||
# tên trường coi là nhạy cảm
|
||||
SENSITIVE = re.compile(
|
||||
r"(api[_-]?key|secret|token|password|passwd|client[_-]?secret|"
|
||||
r"access[_-]?key|private[_-]?key|credential)", re.I)
|
||||
|
||||
# hình dạng credential thật — bắt được kể cả khi tên trường vô hại
|
||||
SHAPES = [
|
||||
("OpenAI", re.compile(r"\bsk-[A-Za-z0-9_\-]{20,}")),
|
||||
("Anthropic", re.compile(r"\bsk-ant-[A-Za-z0-9_\-]{20,}")),
|
||||
("GitHub", re.compile(r"\bgh[pousr]_[A-Za-z0-9]{30,}")),
|
||||
("Slack", re.compile(r"\bxox[abprs]-[A-Za-z0-9\-]{10,}")),
|
||||
("AWS", re.compile(r"\bAKIA[0-9A-Z]{16}\b")),
|
||||
("Google", re.compile(r"\bAIza[0-9A-Za-z_\-]{35}\b")),
|
||||
("JWT", re.compile(r"\beyJ[A-Za-z0-9_\-]{10,}\.[A-Za-z0-9_\-]{10,}\.")),
|
||||
("Private key", re.compile(r"-----BEGIN [A-Z ]*PRIVATE KEY-----")),
|
||||
]
|
||||
|
||||
#: Dòng có dấu này được bỏ qua — lối thoát chuẩn cho mẫu thử, tài liệu, hằng
|
||||
#: đặt tên chứa "secret". Bắt buộc ghi lý do sau dấu hai chấm.
|
||||
ALLOW_MARK = re.compile(r"#\s*casan:\s*allow")
|
||||
|
||||
#: Giá trị là KHOÁ i18n / tên hằng, không phải bí mật. Bắt bằng hình dạng
|
||||
#: "a.b.c" hoặc "a_b_c" chứ không phải bằng danh sách đen từng chữ.
|
||||
LOOKS_LIKE_KEY = re.compile(r"^[a-z][a-z0-9_]*(\.[a-z][a-z0-9_]*)+$")
|
||||
|
||||
#: Credential thật gần như luôn dài hơn thế này. Ngưỡng để loại dữ liệu test
|
||||
#: kiểu api_key="x" — báo động giả làm cả đội thôi đọc báo cáo.
|
||||
MIN_SECRET_LEN = 12
|
||||
|
||||
#: Giá trị là hằng liệt kê, không phải bí mật: mức độ cảnh báo, bật/tắt…
|
||||
ENUMISH = {"warning", "warn", "error", "info", "debug", "critical", "on", "off",
|
||||
"true", "false", "yes", "no", "allow", "deny", "block", "ask",
|
||||
"always", "never", "auto", "default", "disabled", "enabled"}
|
||||
|
||||
# giá trị vô hại — không tính là phát hiện
|
||||
PLACEHOLDER = re.compile(
|
||||
r"^(|ollama|none|null|changeme|your[_\- ]?(api[_\- ]?)?key([_\- ]?here)?|"
|
||||
r"<[^>]*>|\{\{.*\}\}|\$\{.*\}|xxx+|\*+|placeholder|todo|example|test|dummy|"
|
||||
r"sk-\.\.\.|\.\.\.)$", re.I)
|
||||
|
||||
# gán chuỗi trong Python: api_key = "..."
|
||||
PY_ASSIGN = re.compile(
|
||||
r"""["']?(\w*(?:api[_-]?key|secret|token|password|credential)\w*)["']?\s*[:=]\s*"""
|
||||
r"""["']([^"']*)["']""", re.I)
|
||||
|
||||
|
||||
def _is_placeholder(value: str) -> bool:
|
||||
v = value.strip()
|
||||
if PLACEHOLDER.match(v) or v.lower() in ENUMISH:
|
||||
return True
|
||||
if LOOKS_LIKE_KEY.match(v): # "monitoring.action_secret_in_output"
|
||||
return True
|
||||
# quá ngắn để là credential thật
|
||||
return len(v) < MIN_SECRET_LEN
|
||||
|
||||
|
||||
def _walk():
|
||||
for path in REPO.rglob("*"):
|
||||
if not path.is_file():
|
||||
continue
|
||||
if any(part in SKIP_DIRS for part in path.parts):
|
||||
continue
|
||||
if path.suffix in CONFIG_SUFFIX or path.suffix == ".py":
|
||||
yield path
|
||||
|
||||
|
||||
def scan() -> list[dict]:
|
||||
findings: list[dict] = []
|
||||
for path in _walk():
|
||||
try:
|
||||
text = path.read_text(encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
continue
|
||||
rel = path.relative_to(REPO).as_posix()
|
||||
|
||||
for lineno, line in enumerate(text.splitlines(), 1):
|
||||
if ALLOW_MARK.search(line):
|
||||
continue
|
||||
# 1. hình dạng credential thật
|
||||
for label, pattern in SHAPES:
|
||||
m = pattern.search(line)
|
||||
if m:
|
||||
findings.append({
|
||||
"file": rel, "line": lineno, "kind": f"{label} credential",
|
||||
"evidence": m.group(0)[:12] + "…",
|
||||
})
|
||||
|
||||
# 2. trường nhạy cảm có giá trị
|
||||
for m in PY_ASSIGN.finditer(line):
|
||||
field, value = m.group(1), m.group(2)
|
||||
if not SENSITIVE.search(field) or _is_placeholder(value):
|
||||
continue
|
||||
findings.append({
|
||||
"file": rel, "line": lineno,
|
||||
"kind": f"trường '{field}' có giá trị",
|
||||
"evidence": value[:6] + "…" if len(value) > 6 else value,
|
||||
})
|
||||
return findings
|
||||
|
||||
|
||||
def _self_test() -> int:
|
||||
"""Một máy quét không tìm thấy gì chỉ có giá trị nếu chứng minh được nó
|
||||
biết tìm. Cắm mẫu xấu và mẫu vô hại, xem có phân biệt đúng không."""
|
||||
import tempfile
|
||||
|
||||
bad = {
|
||||
"OpenAI": '"api_key": "sk-proj-abc123def456ghi789jkl012mno"', # casan: allow - mau thu cua chinh script
|
||||
"GitHub": 'token = "ghp_ABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789"', # casan: allow - mau thu cua chinh script
|
||||
"AWS": 'aws = "AKIAIOSFODNN7EXAMPLE"', # casan: allow - mau thu cua chinh script
|
||||
"Anthropic": '"api_key": "sk-ant-api03-xxxxxxxxxxxxxxxxxxxxxx"', # casan: allow - mau thu cua chinh script
|
||||
}
|
||||
ok = {
|
||||
"rỗng": '"api_key": ""',
|
||||
"placeholder": '"api_key": "your-key-here"',
|
||||
"ollama": '"api_key": "ollama"',
|
||||
"test ngắn": 'api_key = "x"',
|
||||
"hằng liệt kê": '"secret_in_output": "warning"',
|
||||
}
|
||||
global REPO
|
||||
keep = REPO
|
||||
passed = True
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
REPO = Path(tmp)
|
||||
for label, line in {**bad, **ok}.items():
|
||||
(REPO / "probe.py").write_text(line + "\n", encoding="utf-8")
|
||||
found = bool(scan())
|
||||
want = label in bad
|
||||
mark = "OK " if found == want else "SAI"
|
||||
if found != want:
|
||||
passed = False
|
||||
verb = "bắt được" if found else "bỏ qua"
|
||||
print(f" [{mark}] {label:14} -> {verb}")
|
||||
REPO = keep
|
||||
print()
|
||||
print("Tự kiểm: " + ("script phân biệt đúng." if passed
|
||||
else "*** script phân biệt SAI ***"))
|
||||
return 0 if passed else 1
|
||||
|
||||
|
||||
def main() -> int:
|
||||
ap = argparse.ArgumentParser(description="CASAN Check 1 — quét credential lộ")
|
||||
ap.add_argument("--json", action="store_true", help="in kết quả dạng JSON")
|
||||
ap.add_argument("--self-test", action="store_true",
|
||||
help="cắm credential giả vào file tạm, kiểm script có bắt được")
|
||||
args = ap.parse_args()
|
||||
|
||||
if args.self_test:
|
||||
return _self_test()
|
||||
|
||||
findings = scan()
|
||||
if args.json:
|
||||
print(json.dumps(findings, ensure_ascii=False, indent=2))
|
||||
else:
|
||||
n_files = sum(1 for _ in _walk())
|
||||
print(f"CASAN Check 1 — quét {n_files} file trong {REPO.name}/")
|
||||
if not findings:
|
||||
print("\n0 credential lưu plaintext. PASS.")
|
||||
else:
|
||||
print(f"\n*** {len(findings)} phát hiện ***\n")
|
||||
for f in findings:
|
||||
print(f" {f['file']}:{f['line']}")
|
||||
print(f" {f['kind']} — {f['evidence']}")
|
||||
return 1 if findings else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,239 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""CASAN Check 3 — Clean Architecture Guard (R01-T03).
|
||||
|
||||
Statically walks the AST of every Python file in the pure-Python layers and
|
||||
fails when a file imports something the layer is not allowed to depend on.
|
||||
|
||||
Why AST instead of ``grep``: a regex over source text cannot tell an import
|
||||
apart from the same words appearing inside a docstring, a comment or a string
|
||||
literal (this repo has several docstrings that legitimately mention
|
||||
``PySide6``). ``ast`` sees only real ``import`` / ``from … import`` nodes, so
|
||||
the check has no false positives and needs no ``# noqa`` escape hatches.
|
||||
|
||||
Rules enforced (see docs/architecture/ADR-001-layered-architecture.md):
|
||||
|
||||
* **I1** ``domain/`` and ``application/`` must be 100% pure Python — no Qt.
|
||||
* **I2** ``domain/`` must not import ``application/``, ``infrastructure/``,
|
||||
``presentation/`` or the legacy ``ui/``.
|
||||
* **I3** ``application/`` must not import ``presentation/`` or ``ui/``.
|
||||
|
||||
Usage::
|
||||
|
||||
python scripts/check_imports.py # scan the whole repo
|
||||
python scripts/check_imports.py domain # scan one layer only
|
||||
|
||||
Exit code is 0 when clean and 1 when at least one violation is found, so it
|
||||
can be wired straight into CI / ``scripts/run_quality_gate.py`` (R10-T02).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import ast
|
||||
import sys
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Dict, Iterable, List, Sequence, Tuple
|
||||
|
||||
# Repository root = parent of this scripts/ folder. Everything below is resolved
|
||||
# relative to it so the checker works no matter what the checkout folder is
|
||||
# named or which directory the developer runs it from.
|
||||
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
|
||||
# The distribution package name. Absolute imports may be written either as
|
||||
# ``from cowork_local.ui import x`` or ``from ui import x`` depending on how the
|
||||
# module was reached; we normalise the prefix away so both spellings are caught.
|
||||
PACKAGE_NAME = "cowork_local"
|
||||
|
||||
# Any import whose first dotted segment is one of these is a GUI toolkit.
|
||||
QT_ROOTS = frozenset({"PySide6", "PySide2", "PyQt5", "PyQt6", "shiboken6", "shiboken2"})
|
||||
|
||||
# Per-layer rules: layer directory -> top-level package names it may not import.
|
||||
# Kept as a plain table so adding a layer later is a one-line change and the
|
||||
# rules stay readable next to the ADR they implement.
|
||||
LAYER_RULES: Dict[str, frozenset] = {
|
||||
# I1 + I2: domain is the innermost layer and depends on nothing but stdlib.
|
||||
"domain": frozenset({"application", "infrastructure", "presentation", "ui", "core"}),
|
||||
# I1 + I3: application may use domain, but never anything that draws pixels.
|
||||
"application": frozenset({"presentation", "ui"}),
|
||||
}
|
||||
|
||||
# Directories that are never production code and therefore never scanned.
|
||||
SKIP_DIRS = frozenset({".git", "__pycache__", ".pytest_cache", "tests", "build", "dist"})
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Violation:
|
||||
"""One forbidden import, carrying enough context to fix it without grepping."""
|
||||
|
||||
path: Path
|
||||
line: int
|
||||
imported: str
|
||||
rule: str
|
||||
|
||||
def render(self) -> str:
|
||||
"""Format as ``file:line: message`` — the shape editors turn into a
|
||||
clickable link, so a CI failure lands the developer on the exact line."""
|
||||
rel = self.path.relative_to(REPO_ROOT).as_posix()
|
||||
# ASCII-only on purpose: this line is printed to a console that may run a
|
||||
# legacy code page (cp932 on the team's Windows boxes), where a non-ASCII
|
||||
# dash raises UnicodeEncodeError and would crash the gate on the very
|
||||
# failure path it exists to report.
|
||||
return f"{rel}:{self.line}: imports '{self.imported}' - {self.rule}"
|
||||
|
||||
|
||||
def iter_python_files(layer_dir: Path) -> Iterable[Path]:
|
||||
"""Yield every production ``.py`` file under ``layer_dir``.
|
||||
|
||||
Test files are excluded on purpose: a test for a pure-Python service is
|
||||
allowed to import Qt (an integration test may need a headless widget), and
|
||||
holding tests to the production rule would push people to disable the gate.
|
||||
"""
|
||||
if not layer_dir.is_dir():
|
||||
return
|
||||
for path in sorted(layer_dir.rglob("*.py")):
|
||||
# Reject a path as soon as ANY of its parent folder names is skippable,
|
||||
# which also covers nested __pycache__ inside a sub-package.
|
||||
if any(part in SKIP_DIRS for part in path.parts):
|
||||
continue
|
||||
yield path
|
||||
|
||||
|
||||
def module_parts(path: Path) -> List[str]:
|
||||
"""Dotted package path of ``path`` relative to the repo root, as a list.
|
||||
|
||||
``domain/agents/agent_event.py`` -> ``["domain", "agents", "agent_event"]``
|
||||
``domain/agents/__init__.py`` -> ``["domain", "agents"]``
|
||||
|
||||
Needed to resolve *relative* imports: ``from ..models import X`` inside
|
||||
``domain/agents/foo.py`` really means ``domain.models``, and only the file's
|
||||
own position tells us that.
|
||||
"""
|
||||
rel = path.relative_to(REPO_ROOT)
|
||||
parts = list(rel.parts)
|
||||
if parts[-1] == "__init__.py":
|
||||
parts.pop()
|
||||
else:
|
||||
parts[-1] = parts[-1][: -len(".py")]
|
||||
return parts
|
||||
|
||||
|
||||
def resolve_relative(parts: Sequence[str], level: int, module: str) -> str:
|
||||
"""Turn a relative import into the absolute top-level package it points at.
|
||||
|
||||
``level`` is the number of leading dots. Level 1 means "the package this
|
||||
module lives in", so we drop the module's own name plus ``level - 1``
|
||||
further parents. Returns the FIRST segment of the resolved path, because
|
||||
the rules are expressed in terms of top-level layers.
|
||||
|
||||
Walking off the top of the tree (more dots than there are parents) yields
|
||||
an empty string, which simply never matches a rule — a malformed import
|
||||
like that is a syntax/packaging problem, not an architecture violation.
|
||||
"""
|
||||
base = list(parts[:-1]) # the package containing this module
|
||||
if level > 1:
|
||||
drop = level - 1
|
||||
if drop > len(base):
|
||||
return ""
|
||||
base = base[: len(base) - drop]
|
||||
tail = module.split(".") if module else []
|
||||
resolved = base + tail
|
||||
return resolved[0] if resolved else ""
|
||||
|
||||
|
||||
def top_level(name: str) -> str:
|
||||
"""First dotted segment of an absolute import, with the distribution package
|
||||
prefix stripped so ``cowork_local.ui.chat_panel`` and ``ui.chat_panel`` are
|
||||
treated as the same dependency."""
|
||||
segments = name.split(".")
|
||||
if segments and segments[0] == PACKAGE_NAME:
|
||||
segments = segments[1:]
|
||||
return segments[0] if segments else ""
|
||||
|
||||
|
||||
def imported_roots(tree: ast.AST, parts: Sequence[str]) -> Iterable[Tuple[str, int, str]]:
|
||||
"""Yield ``(top_level_package, line_number, as_written)`` for every import.
|
||||
|
||||
``as_written`` is kept so the error message shows what the developer
|
||||
actually typed rather than the normalised root, which makes the violation
|
||||
obvious at a glance.
|
||||
|
||||
``ast.walk`` (not just the module body) is deliberate: this repo defers many
|
||||
heavy imports into function bodies to keep app start-up fast, and a
|
||||
function-local ``from PySide6 import QtWidgets`` breaks the layer exactly
|
||||
the same way a top-level one does.
|
||||
"""
|
||||
for node in ast.walk(tree):
|
||||
if isinstance(node, ast.Import):
|
||||
for alias in node.names:
|
||||
yield top_level(alias.name), node.lineno, alias.name
|
||||
elif isinstance(node, ast.ImportFrom):
|
||||
if node.level:
|
||||
written = "." * node.level + (node.module or "")
|
||||
yield resolve_relative(parts, node.level, node.module or ""), node.lineno, written
|
||||
else:
|
||||
module = node.module or ""
|
||||
yield top_level(module), node.lineno, module
|
||||
|
||||
|
||||
def check_file(path: Path, layer: str, banned: frozenset) -> List[Violation]:
|
||||
"""Collect every rule violation in one file.
|
||||
|
||||
A file that cannot be parsed is reported as a violation rather than skipped:
|
||||
silently passing a file the checker could not read would make the gate lie.
|
||||
"""
|
||||
try:
|
||||
tree = ast.parse(path.read_text(encoding="utf-8"), filename=str(path))
|
||||
except (SyntaxError, UnicodeDecodeError) as exc:
|
||||
return [Violation(path, getattr(exc, "lineno", 0) or 0, "<unparseable>",
|
||||
f"cannot be parsed by the architecture guard ({exc})")]
|
||||
|
||||
parts = module_parts(path)
|
||||
out: List[Violation] = []
|
||||
for root, lineno, written in imported_roots(tree, parts):
|
||||
if root in QT_ROOTS:
|
||||
out.append(Violation(path, lineno, written,
|
||||
f"'{layer}/' must be 100% pure Python (ADR-001 I1)"))
|
||||
elif root in banned:
|
||||
out.append(Violation(path, lineno, written,
|
||||
f"'{layer}/' must not depend on '{root}/' (ADR-001 I2/I3)"))
|
||||
return out
|
||||
|
||||
|
||||
def run(layers: Sequence[str]) -> List[Violation]:
|
||||
"""Scan the requested layers and return every violation found, in file order."""
|
||||
found: List[Violation] = []
|
||||
for layer in layers:
|
||||
banned = LAYER_RULES[layer]
|
||||
for path in iter_python_files(REPO_ROOT / layer):
|
||||
found.extend(check_file(path, layer, banned))
|
||||
return found
|
||||
|
||||
|
||||
def main(argv: Sequence[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="CASAN Check 3 - Clean Architecture Guard (see ADR-001).")
|
||||
parser.add_argument(
|
||||
"layers", nargs="*", choices=sorted(LAYER_RULES) or None, default=None,
|
||||
help="Layers to scan (default: every layer with a rule).",
|
||||
)
|
||||
args = parser.parse_args(argv)
|
||||
layers = args.layers or sorted(LAYER_RULES)
|
||||
|
||||
violations = run(layers)
|
||||
scanned = sum(1 for layer in layers for _ in iter_python_files(REPO_ROOT / layer))
|
||||
|
||||
if violations:
|
||||
print(f"FAIL - {len(violations)} architecture violation(s) in {scanned} file(s):\n")
|
||||
for v in violations:
|
||||
print(" " + v.render())
|
||||
# Point at the rationale instead of just the rule id, so someone hitting
|
||||
# this for the first time knows where the decision was made.
|
||||
print("\nSee docs/architecture/ADR-001-layered-architecture.md")
|
||||
return 1
|
||||
|
||||
print(f"PASS - 0 Qt imports in {', '.join(layers)} ({scanned} file(s) scanned)")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -52,16 +52,7 @@ class AppContext:
|
||||
# own event loop), so concurrent model calls never needed serializing.
|
||||
self._conn_lock = threading.Lock()
|
||||
self._routing_service = None # lazy RoutingService (Auto Model Routing)
|
||||
# Lazy RoutingApplicationService (R03-T03) — the Qt-free decision layer
|
||||
# every chat surface now routes through. Wraps _routing_service, which
|
||||
# stays the scoring/ranking engine underneath.
|
||||
self._routing_application = None
|
||||
self._routing_lock = threading.Lock()
|
||||
# A SEPARATE lock for the application service: building it calls
|
||||
# routing(), which takes _routing_lock. threading.Lock is not
|
||||
# reentrant, so sharing one lock across both accessors deadlocks the
|
||||
# first caller instead of just serialising them.
|
||||
self._routing_app_lock = threading.Lock()
|
||||
# The workspace (project) currently selected in the Workspace screen.
|
||||
# Per-workspace modes (routing + auto-run) resolve against THIS project
|
||||
# so each workspace keeps its own modes. Updated by WorkspaceTab on
|
||||
@@ -88,28 +79,16 @@ class AppContext:
|
||||
workspace keep its own routing mode."""
|
||||
project = self._current_project()
|
||||
if project is not None:
|
||||
# Validated through the single mode vocabulary (R03-T03) rather
|
||||
# than a literal tuple, so a workspace can store any mode the
|
||||
# routing service understands - including "fallback", whose
|
||||
# on-screen toggle arrives in EPIC R08.
|
||||
from .application.model_routing import is_valid_mode, normalize_mode
|
||||
|
||||
mode = (project.routing_modes or {}).get(surface, "")
|
||||
# Only a RECOGNISED override wins; an empty or corrupt value falls
|
||||
# through to the global setting, exactly as before. Validation goes
|
||||
# through the routing vocabulary (R03-T03) instead of a literal
|
||||
# tuple, so a new mode works everywhere the moment it is defined.
|
||||
if is_valid_mode(mode):
|
||||
return normalize_mode(mode)
|
||||
if mode in ("off", "auto", "manual"):
|
||||
return mode
|
||||
return self.config.routing_mode_for(surface)
|
||||
|
||||
def set_project_routing_mode(self, surface: str, mode: str) -> None:
|
||||
"""Persist a surface's routing mode for the ACTIVE workspace. With no
|
||||
workspace selected, falls back to the global setting so behaviour
|
||||
outside a project stays global."""
|
||||
from .application.model_routing import normalize_mode
|
||||
|
||||
mode = normalize_mode(mode)
|
||||
mode = mode if mode in ("off", "auto", "manual") else "off"
|
||||
project = self._current_project()
|
||||
if project is None:
|
||||
self.config.set_routing_mode_for(surface, mode)
|
||||
@@ -165,34 +144,6 @@ class AppContext:
|
||||
self._routing_service = RoutingService(self)
|
||||
return self._routing_service
|
||||
|
||||
def routing_application(self):
|
||||
"""The shared :class:`RoutingApplicationService` (R03-T03).
|
||||
|
||||
This is what UI code should call: it owns the Off/Auto/Manual/Fallback
|
||||
policy, the confirm handshake and the never-raise guarantee, while
|
||||
:meth:`routing` remains the scoring engine underneath. Chat, Co4E and
|
||||
AI-Edit all go through this one object, so a change to routing policy is
|
||||
made once instead of three times.
|
||||
|
||||
Built lazily and memoised for the same reason as :meth:`routing`: the
|
||||
pending-switch registry and assessment store must be shared app-wide."""
|
||||
if self._routing_application is None:
|
||||
# Resolve the engine BEFORE taking this lock: routing() takes
|
||||
# _routing_lock, and nesting the two acquisitions is what makes the
|
||||
# ordering fragile in the first place.
|
||||
engine = self.routing()
|
||||
with self._routing_app_lock:
|
||||
if self._routing_application is None:
|
||||
from .application.model_routing import RoutingApplicationService
|
||||
|
||||
self._routing_application = RoutingApplicationService(
|
||||
engine,
|
||||
# Per-workspace mode lookup, so each workspace keeps its
|
||||
# own routing behaviour (see project_routing_mode).
|
||||
mode_reader=self.project_routing_mode,
|
||||
)
|
||||
return self._routing_application
|
||||
|
||||
def build_active_provider(self):
|
||||
"""Construct the currently selected provider (called inside workers)."""
|
||||
return self.build_provider_for(self.config.active_provider)
|
||||
|
||||
@@ -1,11 +0,0 @@
|
||||
"""Characterization tests: pin the CURRENT behaviour of legacy code (R01-T04).
|
||||
|
||||
These are not specifications of what the code *should* do - they are a snapshot
|
||||
of what it *does* today, written before the refactor so that any behavioural
|
||||
drift introduced while moving logic into ``application/`` shows up as a failing
|
||||
test rather than as a bug report from a user.
|
||||
|
||||
Rule for this folder: when a test here fails during the refactor, do not "fix"
|
||||
the test first. Decide deliberately whether the behaviour change is intended,
|
||||
and only then update the snapshot in the same commit as the change.
|
||||
"""
|
||||
@@ -1,288 +0,0 @@
|
||||
"""Characterization snapshot of ``core.chat_agent.run_cowork`` (R01-T04).
|
||||
|
||||
``run_cowork`` is the turn engine every Cowork surface funnels through (chat tab,
|
||||
Co4E flow steps, Schedule Task runs). EPIC R04 moves its orchestration into
|
||||
``application/conversations/conversation_application_service.py``; these tests
|
||||
lock down the observable contract BEFORE that move so the new service can be
|
||||
proven equivalent:
|
||||
|
||||
* which system prompt ends up in ``messages``
|
||||
* which tools are advertised to the provider
|
||||
* the exact ``emit`` event sequence for a plain turn and for a tool turn
|
||||
* that ``save_file`` produces a real file in the turn's output folder
|
||||
* that ``cancel`` stops the loop without calling the provider
|
||||
|
||||
Everything runs offline: :class:`FakeProvider` replaces the network and the two
|
||||
disk-backed prompt sources (skills, security rules) are stubbed to empty so the
|
||||
snapshot does not depend on the developer's own ``~/.cowork_local`` contents.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List
|
||||
|
||||
import pytest
|
||||
|
||||
from cowork_local.core import chat_agent
|
||||
from tests.fakes import FakeProvider, FakeToolExecutor, ScriptedTurn
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def isolated_agent(monkeypatch, tmp_path: Path):
|
||||
"""Neutralise every ambient input ``run_cowork`` reads from the machine.
|
||||
|
||||
Without this the snapshot would silently depend on whichever skills and
|
||||
security rules the developer happens to have enabled locally, and on the
|
||||
real audit log under ``~/.cowork_local`` - the test would then pass on one
|
||||
laptop and fail on another for reasons unrelated to the code under test.
|
||||
"""
|
||||
monkeypatch.setattr(chat_agent, "active_skills_text", lambda: "")
|
||||
monkeypatch.setattr(chat_agent, "load_rules", lambda: "")
|
||||
# audit_log is imported lazily inside run_cowork, so patch the module's own
|
||||
# target directory rather than the name chat_agent sees.
|
||||
from cowork_local.core import audit_log
|
||||
|
||||
monkeypatch.setattr(audit_log, "AUDIT_DIR", tmp_path / "audit")
|
||||
return tmp_path
|
||||
|
||||
|
||||
def _run(provider, messages, out_dir: Path, **kwargs):
|
||||
"""Run one turn and return ``(returned_messages, emitted_events)``."""
|
||||
events: List[Dict[str, Any]] = []
|
||||
result = chat_agent.run_cowork(provider, messages, out_dir, events.append, **kwargs)
|
||||
return result, events
|
||||
|
||||
|
||||
def _types(events: List[Dict[str, Any]]) -> List[str]:
|
||||
"""Event ``type`` values in order - the shape assertions read on."""
|
||||
return [e.get("type") for e in events]
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# A plain answer with no tool calls
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_plain_turn_streams_text_and_appends_assistant_message(isolated_agent):
|
||||
out_dir = isolated_agent / "out"
|
||||
provider = FakeProvider([ScriptedTurn(text="Hello there.")])
|
||||
messages: List[Dict[str, Any]] = [{"role": "user", "content": "hi"}]
|
||||
|
||||
result, events = _run(provider, messages, out_dir)
|
||||
|
||||
# The loop ends as soon as the model stops calling tools: exactly one call.
|
||||
assert provider.call_count == 1
|
||||
# run_cowork mutates and returns the SAME list the caller passed in - callers
|
||||
# (ui/cowork_tab.py::build_job) rely on this to persist conversation history.
|
||||
assert result is messages
|
||||
assert result[-1]["role"] == "assistant"
|
||||
assert result[-1]["content"] == "Hello there."
|
||||
assert _types(events) == ["text", "assistant_done"]
|
||||
assert events[0]["delta"] == "Hello there."
|
||||
assert events[-1]["content"] == "Hello there."
|
||||
|
||||
|
||||
def test_system_prompt_is_inserted_once_at_the_front(isolated_agent):
|
||||
out_dir = isolated_agent / "out"
|
||||
provider = FakeProvider([ScriptedTurn(text="ok")])
|
||||
messages: List[Dict[str, Any]] = [{"role": "user", "content": "hi"}]
|
||||
|
||||
result, _ = _run(provider, messages, out_dir)
|
||||
|
||||
assert result[0]["role"] == "system"
|
||||
assert result[0]["content"].startswith("You are Cowork Local")
|
||||
# Exactly one system message: a second turn on the same conversation must not
|
||||
# stack another copy of the prompt (that would grow the context every turn).
|
||||
assert sum(1 for m in result if m.get("role") == "system") == 1
|
||||
|
||||
|
||||
def test_caller_supplied_system_prompt_is_preserved(isolated_agent):
|
||||
"""A caller that already put a system message first keeps its own prompt.
|
||||
|
||||
Co4E flow steps depend on this to give a step its own persona instead of the
|
||||
generic Cowork prompt.
|
||||
"""
|
||||
out_dir = isolated_agent / "out"
|
||||
provider = FakeProvider([ScriptedTurn(text="ok")])
|
||||
messages: List[Dict[str, Any]] = [
|
||||
{"role": "system", "content": "CUSTOM PERSONA"},
|
||||
{"role": "user", "content": "hi"},
|
||||
]
|
||||
|
||||
result, _ = _run(provider, messages, out_dir)
|
||||
|
||||
assert result[0]["content"] == "CUSTOM PERSONA"
|
||||
|
||||
|
||||
def test_reasoning_is_emitted_separately_and_never_joins_the_answer(isolated_agent):
|
||||
"""Reasoning drives the "Thinking" indicator only - it must not become part
|
||||
of the assistant's content, otherwise a reasoning model's private chain of
|
||||
thought would be persisted into conversation history."""
|
||||
out_dir = isolated_agent / "out"
|
||||
provider = FakeProvider([ScriptedTurn(text="42", reasoning="let me think...")])
|
||||
|
||||
result, events = _run(provider, [{"role": "user", "content": "q"}], out_dir)
|
||||
|
||||
assert _types(events) == ["reasoning", "text", "assistant_done"]
|
||||
assert result[-1]["content"] == "42"
|
||||
assert "let me think" not in result[-1]["content"]
|
||||
|
||||
|
||||
def test_reasoning_only_reply_gets_a_placeholder_answer(isolated_agent):
|
||||
"""A model that returns only reasoning must not end the turn on a blank
|
||||
bubble - headless callers (Schedule Task) read this content back as the
|
||||
run's final answer and would otherwise write "(no output)"."""
|
||||
out_dir = isolated_agent / "out"
|
||||
provider = FakeProvider([ScriptedTurn(text="", reasoning="thinking")])
|
||||
|
||||
result, events = _run(provider, [{"role": "user", "content": "q"}], out_dir)
|
||||
|
||||
assert result[-1]["content"].startswith("*(model returned only its reasoning")
|
||||
assert "text" in _types(events)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Tool advertising
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_save_file_and_update_plan_are_always_advertised(isolated_agent):
|
||||
out_dir = isolated_agent / "out"
|
||||
provider = FakeProvider([ScriptedTurn(text="ok")])
|
||||
|
||||
_run(provider, [{"role": "user", "content": "hi"}], out_dir)
|
||||
|
||||
advertised = provider.calls[0].tool_names
|
||||
assert "save_file" in advertised
|
||||
assert "update_plan" in advertised
|
||||
|
||||
|
||||
def test_allowed_tools_scopes_the_catalogue_but_keeps_update_plan(isolated_agent):
|
||||
"""``allowed_tools`` is the permission scope Co4E steps use: a read-only step
|
||||
must literally not be offered a writing tool. ``update_plan`` survives the
|
||||
filter because it has no side effects."""
|
||||
out_dir = isolated_agent / "out"
|
||||
provider = FakeProvider([ScriptedTurn(text="ok")])
|
||||
|
||||
_run(provider, [{"role": "user", "content": "hi"}], out_dir,
|
||||
allowed_tools=["read_file"])
|
||||
|
||||
advertised = set(provider.calls[0].tool_names)
|
||||
assert "save_file" not in advertised
|
||||
assert "update_plan" in advertised
|
||||
|
||||
|
||||
def test_extra_tools_are_advertised_alongside_built_ins(isolated_agent):
|
||||
out_dir = isolated_agent / "out"
|
||||
executor = FakeToolExecutor(results={"ms365_send_mail": {"output": "sent"}})
|
||||
provider = FakeProvider([ScriptedTurn(text="ok")])
|
||||
|
||||
_run(provider, [{"role": "user", "content": "hi"}], out_dir,
|
||||
extra_tools=executor.specs(), extra_executor=executor)
|
||||
|
||||
assert "ms365_send_mail" in provider.calls[0].tool_names
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Tool execution
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_save_file_writes_a_real_file_and_reports_it(isolated_agent):
|
||||
out_dir = isolated_agent / "out"
|
||||
provider = FakeProvider([
|
||||
ScriptedTurn(tool_calls=[("save_file", {"filename": "note.md",
|
||||
"content": "# Result\n"})]),
|
||||
ScriptedTurn(text="Done."),
|
||||
])
|
||||
|
||||
result, events = _run(provider, [{"role": "user", "content": "make a note"}], out_dir)
|
||||
|
||||
written = [p for p in out_dir.iterdir() if p.is_file()]
|
||||
assert len(written) == 1
|
||||
assert written[0].read_text(encoding="utf-8") == "# Result\n"
|
||||
|
||||
assert _types(events) == [
|
||||
"assistant_done", # first turn: tool call only, no visible text
|
||||
"tool_proposed", # the diff preview shown in the chat
|
||||
"tool_result",
|
||||
"text", # second turn's answer
|
||||
"assistant_done",
|
||||
]
|
||||
assert events[2]["ok"] is True
|
||||
|
||||
# The tool result is fed back as a `tool` message so the model can react to it.
|
||||
roles = [m["role"] for m in result]
|
||||
assert roles == ["system", "user", "assistant", "tool", "assistant"]
|
||||
assert result[3]["name"] == "save_file"
|
||||
|
||||
|
||||
def test_extra_tool_calls_are_routed_to_the_extra_executor(isolated_agent):
|
||||
"""MCP / Microsoft 365 tools bypass the built-in file+command handlers and go
|
||||
to the caller-supplied executor instead."""
|
||||
out_dir = isolated_agent / "out"
|
||||
executor = FakeToolExecutor(results={"ms365_send_mail": {"ok": True, "output": "sent"}})
|
||||
provider = FakeProvider([
|
||||
ScriptedTurn(tool_calls=[("ms365_send_mail", {"to": "a@b.c"})]),
|
||||
ScriptedTurn(text="Mail sent."),
|
||||
])
|
||||
|
||||
result, events = _run(provider, [{"role": "user", "content": "mail them"}], out_dir,
|
||||
extra_tools=executor.specs(), extra_executor=executor)
|
||||
|
||||
assert executor.call_names == ["ms365_send_mail"]
|
||||
assert executor.args_for("ms365_send_mail") == [{"to": "a@b.c"}]
|
||||
assert [e for e in events if e["type"] == "tool_result"][0]["output"] == "sent"
|
||||
assert result[3] == {"role": "tool", "tool_call_id": result[3]["tool_call_id"],
|
||||
"name": "ms365_send_mail", "content": "sent"}
|
||||
|
||||
|
||||
def test_update_plan_drives_the_plan_panel_without_producing_a_file(isolated_agent):
|
||||
out_dir = isolated_agent / "out"
|
||||
provider = FakeProvider([
|
||||
ScriptedTurn(tool_calls=[("update_plan", {"steps": [{"title": "step one"}]})]),
|
||||
ScriptedTurn(text="Planned."),
|
||||
])
|
||||
|
||||
result, events = _run(provider, [{"role": "user", "content": "plan it"}], out_dir)
|
||||
|
||||
plan_events = [e for e in events if e["type"] == "plan_set"]
|
||||
assert len(plan_events) == 1
|
||||
assert plan_events[0]["steps"]
|
||||
# No tool_proposed/tool_result bubbles for a plan update, and no file on disk.
|
||||
assert "tool_proposed" not in _types(events)
|
||||
assert list(out_dir.iterdir()) == []
|
||||
assert result[3]["content"] == "Plan updated."
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Cancellation
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_cancel_before_the_first_step_never_calls_the_provider(isolated_agent):
|
||||
"""Stop pressed before the loop starts must cost zero tokens."""
|
||||
out_dir = isolated_agent / "out"
|
||||
provider = FakeProvider([], strict=True)
|
||||
|
||||
result, events = _run(provider, [{"role": "user", "content": "hi"}], out_dir,
|
||||
cancel=lambda: True)
|
||||
|
||||
assert provider.call_count == 0
|
||||
assert _types(events) == []
|
||||
# The system prompt is still installed, so the conversation stays well-formed
|
||||
# for a later retry on the same message list.
|
||||
assert result[0]["role"] == "system"
|
||||
|
||||
|
||||
def test_cancel_between_steps_stops_before_the_next_provider_call(isolated_agent):
|
||||
"""After a tool call runs, a Stop must end the turn instead of paying for
|
||||
another round trip."""
|
||||
out_dir = isolated_agent / "out"
|
||||
provider = FakeProvider([
|
||||
ScriptedTurn(tool_calls=[("save_file", {"filename": "a.md", "content": "x"})]),
|
||||
])
|
||||
calls = {"n": 0}
|
||||
|
||||
def cancel() -> bool:
|
||||
# False on the first check (loop entry), True afterwards - i.e. the user
|
||||
# pressed Stop while the first step was running.
|
||||
calls["n"] += 1
|
||||
return calls["n"] > 1
|
||||
|
||||
result, _ = _run(provider, [{"role": "user", "content": "hi"}], out_dir, cancel=cancel)
|
||||
|
||||
assert provider.call_count == 1
|
||||
assert result[-1]["role"] in {"assistant", "tool"}
|
||||
+4
-57
@@ -1,63 +1,10 @@
|
||||
"""Root pytest configuration: bind ``cowork_local`` to THIS checkout (R01-T02).
|
||||
"""Make the repository package importable when pytest runs from the repo root."""
|
||||
|
||||
Why this file exists
|
||||
--------------------
|
||||
The package directory is itself the distribution package (``__init__.py`` sits
|
||||
at the repo root), so ``import cowork_local`` only resolves when the checkout
|
||||
folder happens to be named exactly ``cowork_local``. It frequently is not — this
|
||||
one is checked out as ``cowork_local_gitea``, and developers keep several dated
|
||||
copies side by side (``cowork_local``, ``cowork_local_20260722``, ...).
|
||||
|
||||
Left alone, ``sys.path``-based discovery would import whichever *sibling* folder
|
||||
is named ``cowork_local`` and the whole suite would silently test a DIFFERENT
|
||||
checkout: green here, broken in the branch under review. That is the worst kind
|
||||
of test failure, because it fails to fail.
|
||||
|
||||
So instead of relying on the folder name, we load ``__init__.py`` by absolute
|
||||
path and register the result in ``sys.modules`` under the canonical name before
|
||||
any test imports it. Submodules (``cowork_local.providers.base``, ...) then
|
||||
resolve through this package's own ``__path__``, i.e. always this checkout.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
# .../<checkout>/tests/conftest.py -> .../<checkout>
|
||||
_PKG_DIR = Path(__file__).resolve().parents[1]
|
||||
_PKG_NAME = "cowork_local"
|
||||
|
||||
|
||||
def _bind_package_to_this_checkout() -> None:
|
||||
"""Make ``import cowork_local`` mean this directory, whatever it is named.
|
||||
|
||||
A no-op when the correct package object is already bound, so running the
|
||||
suite from a folder that IS named ``cowork_local`` costs nothing and the
|
||||
hook stays idempotent across repeated conftest collection.
|
||||
"""
|
||||
existing = sys.modules.get(_PKG_NAME)
|
||||
existing_file = getattr(existing, "__file__", None)
|
||||
if existing_file and Path(existing_file).resolve().parent == _PKG_DIR:
|
||||
return # already the right one
|
||||
|
||||
spec = importlib.util.spec_from_file_location(
|
||||
_PKG_NAME,
|
||||
_PKG_DIR / "__init__.py",
|
||||
# Setting the search locations is what makes dotted submodule imports
|
||||
# (cowork_local.core.*, cowork_local.providers.*) resolve inside THIS
|
||||
# directory rather than through sys.path.
|
||||
submodule_search_locations=[str(_PKG_DIR)],
|
||||
)
|
||||
if spec is None or spec.loader is None: # pragma: no cover - packaging error
|
||||
raise RuntimeError(f"cannot load {_PKG_NAME} from {_PKG_DIR}")
|
||||
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
# Registered BEFORE exec_module so that a self-referential import inside
|
||||
# __init__.py would find the partially-initialised module instead of
|
||||
# recursing - the same protocol CPython's own import machinery follows.
|
||||
sys.modules[_PKG_NAME] = module
|
||||
spec.loader.exec_module(module)
|
||||
|
||||
|
||||
_bind_package_to_this_checkout()
|
||||
REPOSITORY_PARENT = Path(__file__).resolve().parents[2]
|
||||
if str(REPOSITORY_PARENT) not in sys.path:
|
||||
sys.path.insert(0, str(REPOSITORY_PARENT))
|
||||
|
||||
@@ -1,8 +0,0 @@
|
||||
"""Contract tests: one shared behaviour suite every implementation must satisfy.
|
||||
|
||||
Unlike unit tests (which test one module in isolation) a contract test is
|
||||
parametrised over EVERY implementation of an interface, so a newly added
|
||||
provider either satisfies the same promises as the existing ones or the suite
|
||||
goes red on the day it is added - not months later, in production, on the one
|
||||
code path that assumed the promise held.
|
||||
"""
|
||||
@@ -1,342 +0,0 @@
|
||||
"""Provider contract suite (R03-T01).
|
||||
|
||||
Every provider - the two real adapters and the test double - must honour the
|
||||
same promises declared in ``providers/base.py``:
|
||||
|
||||
1. ``chat()`` returns the canonical assistant message
|
||||
``{"role": "assistant", "content": str, "tool_calls": [...]}``.
|
||||
2. Answer text is streamed through ``on_text`` and equals the returned content.
|
||||
3. Private reasoning goes to ``on_reasoning`` ONLY - it must never leak into the
|
||||
answer, or a reasoning model's chain of thought ends up persisted in history.
|
||||
4. Tool calls come back as ``{"id", "name", "arguments": dict}`` with arguments
|
||||
already parsed - callers must never have to json.loads() them.
|
||||
5. A failure raises ``ProviderError`` and nothing else, so one except clause in
|
||||
the agent loop covers every provider.
|
||||
|
||||
The real adapters are exercised WITHOUT network access by replacing
|
||||
``Provider._request`` with a canned SSE response - which is exactly the seam
|
||||
``providers/base.py`` documents for its TLS retry, so no production code needed
|
||||
changing to make this testable.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
import pytest
|
||||
|
||||
from cowork_local.domain.models.provider_descriptor import ProviderCapability
|
||||
from cowork_local.infrastructure.providers.provider_registry import (
|
||||
BUILT_IN_PROVIDERS,
|
||||
ProviderRegistry,
|
||||
)
|
||||
from cowork_local.providers.anthropic import AnthropicProvider
|
||||
from cowork_local.providers.base import Provider, ProviderError, ToolSpec
|
||||
from cowork_local.providers.openai_compat import OpenAICompatProvider
|
||||
from tests.fakes import FakeProvider, ScriptedTurn
|
||||
|
||||
|
||||
class _StubResponse:
|
||||
"""Minimal stand-in for a streamed ``requests.Response``.
|
||||
|
||||
Only the members the provider code actually touches are implemented; adding
|
||||
more would invite tests that pass against the stub but not against requests.
|
||||
"""
|
||||
|
||||
def __init__(self, lines: List[str], status_code: int = 200, text: str = "") -> None:
|
||||
self._lines = lines
|
||||
self.status_code = status_code
|
||||
self.text = text
|
||||
self.headers: Dict[str, str] = {}
|
||||
self.encoding = "utf-8"
|
||||
self.closed = False
|
||||
|
||||
def iter_lines(self, decode_unicode: bool = False):
|
||||
yield from self._lines
|
||||
|
||||
def close(self) -> None:
|
||||
self.closed = True
|
||||
|
||||
def json(self) -> Any:
|
||||
return json.loads(self.text or "{}")
|
||||
|
||||
|
||||
def _sse(*payloads: Dict[str, Any]) -> List[str]:
|
||||
"""Render payloads as SSE ``data:`` lines, the wire shape both adapters parse."""
|
||||
return [f"data: {json.dumps(p)}" for p in payloads]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def canned(monkeypatch):
|
||||
"""Return a helper that makes every provider request answer with ``lines``."""
|
||||
|
||||
def _install(lines: List[str], status_code: int = 200, text: str = "") -> Dict[str, Any]:
|
||||
seen: Dict[str, Any] = {}
|
||||
|
||||
def fake_request(self, method, url, **kwargs):
|
||||
# Capture the outgoing payload so tests can assert on how the
|
||||
# canonical message list was translated to the provider's wire format.
|
||||
seen["method"] = method
|
||||
seen["url"] = url
|
||||
seen["json"] = kwargs.get("json")
|
||||
return _StubResponse(lines, status_code=status_code, text=text)
|
||||
|
||||
monkeypatch.setattr(Provider, "_request", fake_request, raising=True)
|
||||
return seen
|
||||
|
||||
return _install
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Shared base-class behaviour every provider inherits
|
||||
# --------------------------------------------------------------------------- #
|
||||
def _providers_under_test() -> List[Provider]:
|
||||
"""One instance of each implementation, configured but never called."""
|
||||
conf = {"base_url": "https://example.invalid/v1", "api_key": "k", "model": "m"}
|
||||
return [
|
||||
OpenAICompatProvider(dict(conf)),
|
||||
AnthropicProvider(dict(conf)),
|
||||
FakeProvider(),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("provider", _providers_under_test(), ids=lambda p: type(p).__name__)
|
||||
def test_every_provider_exposes_the_base_contract(provider):
|
||||
assert isinstance(provider, Provider)
|
||||
assert callable(provider.chat)
|
||||
assert callable(provider.list_models)
|
||||
assert callable(provider.test_connection)
|
||||
# `name` identifies the provider in usage records and audit entries; an
|
||||
# implementation that forgot to set it would silently report as "base".
|
||||
assert provider.name and provider.name != "base"
|
||||
assert isinstance(provider.supports_vision, bool)
|
||||
assert provider.describe() == f"{provider.name}:{provider.model}"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("provider", _providers_under_test(), ids=lambda p: type(p).__name__)
|
||||
def test_strip_think_removes_inline_reasoning_from_a_final_answer(provider):
|
||||
"""Safety net for gateways that fold reasoning into the content stream: the
|
||||
answer stored in history must never contain a <think> block."""
|
||||
assert provider.strip_think("<think>secret</think>Answer") == "Answer"
|
||||
assert provider.strip_think("Plain answer") == "Plain answer"
|
||||
|
||||
|
||||
def test_tool_spec_translates_to_both_wire_formats():
|
||||
"""One ToolSpec must render for both protocols - this is what lets the agent
|
||||
loop build its tool catalogue once and reuse it across providers."""
|
||||
spec = ToolSpec(name="save_file", description="Write a file",
|
||||
parameters={"type": "object", "properties": {}})
|
||||
|
||||
openai_shape = spec.to_openai()
|
||||
anthropic_shape = spec.to_anthropic()
|
||||
|
||||
assert openai_shape["type"] == "function"
|
||||
assert openai_shape["function"]["name"] == "save_file"
|
||||
assert openai_shape["function"]["parameters"] == spec.parameters
|
||||
# Anthropic names the same field `input_schema`; the values must stay equal,
|
||||
# otherwise the same tool would validate differently per provider.
|
||||
assert anthropic_shape["name"] == "save_file"
|
||||
assert anthropic_shape["input_schema"] == spec.parameters
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Streaming contract - real adapters, canned transport
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_openai_compat_streams_text_and_returns_canonical_message(canned):
|
||||
canned(_sse(
|
||||
{"choices": [{"delta": {"content": "Hel"}}]},
|
||||
{"choices": [{"delta": {"content": "lo"}}]},
|
||||
) + ["data: [DONE]"])
|
||||
provider = OpenAICompatProvider({"base_url": "https://x.invalid/v1",
|
||||
"api_key": "k", "model": "m"})
|
||||
chunks: List[str] = []
|
||||
|
||||
result = provider.chat([{"role": "user", "content": "hi"}], on_text=chunks.append)
|
||||
|
||||
assert "".join(chunks) == "Hello"
|
||||
assert result["role"] == "assistant"
|
||||
assert result["content"] == "Hello"
|
||||
assert result["tool_calls"] == []
|
||||
|
||||
|
||||
def test_openai_compat_keeps_reasoning_out_of_the_answer(canned):
|
||||
canned(_sse(
|
||||
{"choices": [{"delta": {"reasoning_content": "hmm..."}}]},
|
||||
{"choices": [{"delta": {"content": "42"}}]},
|
||||
) + ["data: [DONE]"])
|
||||
provider = OpenAICompatProvider({"base_url": "https://x.invalid/v1",
|
||||
"api_key": "k", "model": "m"})
|
||||
text: List[str] = []
|
||||
reasoning: List[str] = []
|
||||
|
||||
result = provider.chat([{"role": "user", "content": "q"}],
|
||||
on_text=text.append, on_reasoning=reasoning.append)
|
||||
|
||||
assert reasoning == ["hmm..."]
|
||||
assert result["content"] == "42"
|
||||
assert "hmm" not in result["content"]
|
||||
|
||||
|
||||
def test_openai_compat_returns_tool_calls_with_parsed_arguments(canned):
|
||||
"""Arguments arrive as a JSON string split across chunks; the contract says
|
||||
the caller receives a ready-to-use dict."""
|
||||
canned(_sse(
|
||||
{"choices": [{"delta": {"tool_calls": [
|
||||
{"index": 0, "id": "call_a", "function": {"name": "save_file",
|
||||
"arguments": '{"filename":'}}]}}]},
|
||||
{"choices": [{"delta": {"tool_calls": [
|
||||
{"index": 0, "function": {"arguments": '"a.md"}'}}]}}]},
|
||||
) + ["data: [DONE]"])
|
||||
provider = OpenAICompatProvider({"base_url": "https://x.invalid/v1",
|
||||
"api_key": "k", "model": "m"})
|
||||
|
||||
result = provider.chat([{"role": "user", "content": "save it"}])
|
||||
|
||||
assert len(result["tool_calls"]) == 1
|
||||
call = result["tool_calls"][0]
|
||||
assert call["id"] == "call_a"
|
||||
assert call["name"] == "save_file"
|
||||
assert call["arguments"] == {"filename": "a.md"}
|
||||
|
||||
|
||||
def test_anthropic_streams_text_and_returns_canonical_message(canned):
|
||||
canned(_sse(
|
||||
{"type": "content_block_delta", "index": 0,
|
||||
"delta": {"type": "text_delta", "text": "Hel"}},
|
||||
{"type": "content_block_delta", "index": 0,
|
||||
"delta": {"type": "text_delta", "text": "lo"}},
|
||||
{"type": "message_stop"},
|
||||
))
|
||||
provider = AnthropicProvider({"base_url": "https://x.invalid",
|
||||
"api_key": "k", "model": "m"})
|
||||
chunks: List[str] = []
|
||||
|
||||
result = provider.chat([{"role": "user", "content": "hi"}], on_text=chunks.append)
|
||||
|
||||
assert "".join(chunks) == "Hello"
|
||||
assert result["content"] == "Hello"
|
||||
assert result["role"] == "assistant"
|
||||
|
||||
|
||||
def test_anthropic_keeps_extended_thinking_out_of_the_answer(canned):
|
||||
canned(_sse(
|
||||
{"type": "content_block_delta", "index": 0,
|
||||
"delta": {"type": "thinking_delta", "thinking": "reasoning..."}},
|
||||
{"type": "content_block_delta", "index": 0,
|
||||
"delta": {"type": "text_delta", "text": "42"}},
|
||||
{"type": "message_stop"},
|
||||
))
|
||||
provider = AnthropicProvider({"base_url": "https://x.invalid",
|
||||
"api_key": "k", "model": "m"})
|
||||
reasoning: List[str] = []
|
||||
|
||||
result = provider.chat([{"role": "user", "content": "q"}], on_reasoning=reasoning.append)
|
||||
|
||||
assert reasoning == ["reasoning..."]
|
||||
assert result["content"] == "42"
|
||||
|
||||
|
||||
def test_anthropic_returns_tool_calls_with_parsed_arguments(canned):
|
||||
canned(_sse(
|
||||
{"type": "content_block_start", "index": 0,
|
||||
"content_block": {"type": "tool_use", "id": "toolu_1", "name": "save_file"}},
|
||||
{"type": "content_block_delta", "index": 0,
|
||||
"delta": {"type": "input_json_delta", "partial_json": '{"filename":"a.md"}'}},
|
||||
{"type": "message_stop"},
|
||||
))
|
||||
provider = AnthropicProvider({"base_url": "https://x.invalid",
|
||||
"api_key": "k", "model": "m"})
|
||||
|
||||
result = provider.chat([{"role": "user", "content": "save"}])
|
||||
|
||||
assert result["tool_calls"] == [
|
||||
{"id": "toolu_1", "name": "save_file", "arguments": {"filename": "a.md"}}
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("factory", [
|
||||
lambda: OpenAICompatProvider({"base_url": "https://x.invalid/v1", "api_key": "k", "model": "m"}),
|
||||
lambda: AnthropicProvider({"base_url": "https://x.invalid", "api_key": "k", "model": "m"}),
|
||||
], ids=["openai_compat", "anthropic"])
|
||||
def test_transport_failure_surfaces_as_provider_error(canned, factory):
|
||||
"""Every failure mode must arrive as ProviderError so the agent loop needs
|
||||
exactly one except clause, whichever provider is active."""
|
||||
canned([], status_code=500, text="boom")
|
||||
|
||||
with pytest.raises(ProviderError):
|
||||
factory().chat([{"role": "user", "content": "hi"}])
|
||||
|
||||
|
||||
def test_fake_provider_satisfies_the_same_streaming_contract():
|
||||
"""The double is only useful as a stand-in if it keeps the same promises the
|
||||
real adapters are held to above."""
|
||||
provider = FakeProvider([ScriptedTurn(text="Hello", reasoning="hmm")])
|
||||
text: List[str] = []
|
||||
reasoning: List[str] = []
|
||||
|
||||
result = provider.chat([{"role": "user", "content": "hi"}],
|
||||
on_text=text.append, on_reasoning=reasoning.append)
|
||||
|
||||
assert "".join(text) == result["content"] == "Hello"
|
||||
assert reasoning == ["hmm"]
|
||||
assert result["role"] == "assistant"
|
||||
assert result["tool_calls"] == []
|
||||
|
||||
|
||||
def test_fake_provider_raises_provider_error_like_the_real_ones():
|
||||
provider = FakeProvider([ScriptedTurn(error="gateway exploded")])
|
||||
|
||||
with pytest.raises(ProviderError):
|
||||
provider.chat([{"role": "user", "content": "hi"}])
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Registry <-> implementation agreement
|
||||
# --------------------------------------------------------------------------- #
|
||||
@pytest.mark.parametrize("descriptor", BUILT_IN_PROVIDERS, ids=lambda d: d.id)
|
||||
def test_every_descriptor_builds_a_working_provider(descriptor):
|
||||
"""A descriptor that cannot be built is a catalogue lying to the UI: Settings
|
||||
would list the provider and selecting it would fail at the first message."""
|
||||
registry = ProviderRegistry()
|
||||
conf = {"base_url": "https://x.invalid/v1", "api_key": "k"}
|
||||
|
||||
provider = registry.build(descriptor.id, conf)
|
||||
|
||||
assert isinstance(provider, Provider)
|
||||
# The id, not the shared adapter class name: three descriptors map onto
|
||||
# OpenAICompatProvider, and usage/audit records must still tell them apart.
|
||||
assert provider.name == descriptor.id
|
||||
assert provider.model == descriptor.default_model
|
||||
|
||||
|
||||
@pytest.mark.parametrize("descriptor", BUILT_IN_PROVIDERS, ids=lambda d: d.id)
|
||||
def test_declared_vision_capability_matches_the_implementation(descriptor):
|
||||
"""``supports_vision`` decides whether an image block may be sent. A
|
||||
descriptor claiming vision for an adapter that cannot translate the block
|
||||
would route image turns into a guaranteed failure."""
|
||||
provider = ProviderRegistry().build(descriptor.id, {"base_url": "u", "api_key": "k"})
|
||||
|
||||
if descriptor.supports(ProviderCapability.VISION):
|
||||
assert provider.supports_vision is True
|
||||
|
||||
|
||||
def test_registry_build_never_mutates_the_caller_config():
|
||||
"""The routing layer runs one turn on a different model; if build() wrote
|
||||
that model back into the config dict it was handed, the override would
|
||||
silently become the user's saved default."""
|
||||
registry = ProviderRegistry()
|
||||
conf = {"base_url": "u", "api_key": "k", "model": "configured-model"}
|
||||
|
||||
provider = registry.build("openai_compat", conf, model="routed-model")
|
||||
|
||||
assert provider.model == "routed-model"
|
||||
assert conf["model"] == "configured-model"
|
||||
|
||||
|
||||
def test_registry_rejects_an_unknown_provider_with_provider_error():
|
||||
with pytest.raises(ProviderError) as excinfo:
|
||||
ProviderRegistry().build("does_not_exist", {})
|
||||
|
||||
# The message lists what IS known, so a typo in config is fixable from the
|
||||
# error alone without opening the source.
|
||||
assert "openai_compat" in str(excinfo.value)
|
||||
+1
-16
@@ -1,16 +1 @@
|
||||
"""Offline test doubles for the refactoring safety net (R01-T02).
|
||||
|
||||
Every double here is deliberately Qt-free, network-free and disk-free so the
|
||||
unit/contract suites run in well under a second and give the same answer on a
|
||||
laptop, in CI and on a machine with no API keys configured.
|
||||
|
||||
* :class:`~tests.fakes.fake_provider.FakeProvider` - a scripted
|
||||
``providers.base.Provider`` that streams canned text/tool calls.
|
||||
* :class:`~tests.fakes.fake_tool_executor.FakeToolExecutor` - a scripted stand-in
|
||||
for the ``extra_executor`` callable that ``core.chat_agent.run_cowork`` routes
|
||||
MCP/connector tool calls to.
|
||||
"""
|
||||
from .fake_provider import FakeProvider, ScriptedTurn
|
||||
from .fake_tool_executor import FakeToolExecutor, ToolInvocation
|
||||
|
||||
__all__ = ["FakeProvider", "ScriptedTurn", "FakeToolExecutor", "ToolInvocation"]
|
||||
"""Test double dùng chung cho cả 3 team — không phụ thuộc Qt."""
|
||||
|
||||
@@ -0,0 +1,138 @@
|
||||
"""Bản giả của ConfigRepository và SecretStore — chạy trong bộ nhớ.
|
||||
|
||||
Dùng để N2 (Giám sát) và N3 (Co4E) code và test ngay từ 21/08, không phải đợi
|
||||
bản thật xong ngày 23/08 và 26/08.
|
||||
|
||||
Không chạm đĩa, không chạm keyring, không cần Qt. Test dùng nó chạy trong vài
|
||||
mili giây.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict
|
||||
|
||||
|
||||
class FakeSecretStore:
|
||||
"""SecretStore trong bộ nhớ.
|
||||
|
||||
>>> s = FakeSecretStore({"provider:openai": "sk-test"})
|
||||
>>> s.get("provider:openai")
|
||||
'sk-test'
|
||||
>>> s.get("provider:chua-co") is None
|
||||
True
|
||||
"""
|
||||
|
||||
def __init__(self, seed: Dict[str, str] | None = None):
|
||||
self._items: Dict[str, str] = dict(seed or {})
|
||||
|
||||
def get(self, key: str) -> str | None:
|
||||
return self._items.get(key)
|
||||
|
||||
def set(self, key: str, value: str) -> None:
|
||||
self._items[key] = value
|
||||
|
||||
def delete(self, key: str) -> None:
|
||||
self._items.pop(key, None)
|
||||
|
||||
def has(self, key: str) -> bool:
|
||||
return key in self._items
|
||||
|
||||
|
||||
class FakeConfigRepository:
|
||||
"""ConfigRepository trong bộ nhớ, có sẵn giá trị mặc định hợp lý.
|
||||
|
||||
Mọi thứ ghi đè được qua tham số khởi tạo, nên test dựng đúng tình huống
|
||||
mình cần::
|
||||
|
||||
cfg = FakeConfigRepository(theme="light", shared_dir="/tmp/chung")
|
||||
"""
|
||||
|
||||
def __init__(self, *, active_provider: str = "ollama",
|
||||
providers: Dict[str, Dict[str, Any]] | None = None,
|
||||
shared_dir: str = "", theme: str = "dark", language: str = "vi",
|
||||
routing: Dict[str, Any] | None = None,
|
||||
auth: Dict[str, Any] | None = None,
|
||||
agent_security: Dict[str, Any] | None = None,
|
||||
tools_disabled: list[str] | None = None,
|
||||
history_dir: Path | None = None,
|
||||
output_dir: Path | None = None):
|
||||
self._active_provider = active_provider
|
||||
self._providers = providers or {
|
||||
"ollama": {"base_url": "http://localhost:11434/v1", "model": "llama3"},
|
||||
"openai": {"base_url": "https://api.openai.com/v1", "model": "gpt-4o-mini"},
|
||||
}
|
||||
self._shared_dir = shared_dir
|
||||
self._theme = theme
|
||||
self._language = language
|
||||
self._routing = routing or {"mode": "off"}
|
||||
self._auth = auth or {}
|
||||
self._agent_security = agent_security or {"cowork_confirm_commands": True}
|
||||
self._tools_disabled = list(tools_disabled or [])
|
||||
self._history_dir = history_dir or Path("/fake/history")
|
||||
self._output_dir = output_dir or Path("/fake/workspace")
|
||||
#: số lần save() được gọi — để test khẳng định "có ghi" mà không cần đĩa
|
||||
self.saves = 0
|
||||
|
||||
# ---- provider ------------------------------------------------------
|
||||
@property
|
||||
def active_provider(self) -> str:
|
||||
return self._active_provider
|
||||
|
||||
def set_active_provider(self, name: str) -> None:
|
||||
self._active_provider = name
|
||||
|
||||
def provider_conf(self, name: str | None = None) -> Dict[str, Any]:
|
||||
return dict(self._providers.get(name or self._active_provider, {}))
|
||||
|
||||
# ---- đường dẫn -----------------------------------------------------
|
||||
@property
|
||||
def shared_dir(self) -> str:
|
||||
return self._shared_dir
|
||||
|
||||
def history_dir(self) -> Path:
|
||||
return self._history_dir
|
||||
|
||||
def cowork_output_dir(self) -> Path:
|
||||
return self._output_dir
|
||||
|
||||
# ---- giao diện -----------------------------------------------------
|
||||
@property
|
||||
def theme(self) -> str:
|
||||
return self._theme
|
||||
|
||||
def set_theme(self, value: str) -> None:
|
||||
self._theme = value
|
||||
|
||||
@property
|
||||
def language(self) -> str:
|
||||
return self._language
|
||||
|
||||
def set_language(self, value: str) -> None:
|
||||
self._language = value
|
||||
|
||||
# ---- nhóm cấu hình --------------------------------------------------
|
||||
@property
|
||||
def routing(self) -> Dict[str, Any]:
|
||||
return self._routing
|
||||
|
||||
@property
|
||||
def auth(self) -> Dict[str, Any]:
|
||||
return self._auth
|
||||
|
||||
@property
|
||||
def agent_security(self) -> Dict[str, Any]:
|
||||
return self._agent_security
|
||||
|
||||
@property
|
||||
def tools_disabled(self) -> list[str]:
|
||||
return list(self._tools_disabled)
|
||||
|
||||
def set_tool_enabled(self, name: str, enabled: bool) -> None:
|
||||
if enabled:
|
||||
self._tools_disabled = [t for t in self._tools_disabled if t != name]
|
||||
elif name not in self._tools_disabled:
|
||||
self._tools_disabled.append(name)
|
||||
|
||||
# ---- ghi ------------------------------------------------------------
|
||||
def save(self) -> None:
|
||||
self.saves += 1
|
||||
@@ -1,213 +0,0 @@
|
||||
"""FakeProvider - a scripted, offline stand-in for a real LLM provider (R01-T02).
|
||||
|
||||
The real providers (``providers/openai_compat.py``, ``providers/anthropic.py``)
|
||||
open HTTP connections, need API keys and stream at the mercy of the network, so
|
||||
nothing above them could be tested deterministically. This double implements the
|
||||
same :class:`providers.base.Provider` contract from a list of scripted turns:
|
||||
|
||||
provider = FakeProvider([
|
||||
ScriptedTurn(tool_calls=[("save_file", {"filename": "a.md", "content": "hi"})]),
|
||||
ScriptedTurn(text="Saved it."),
|
||||
])
|
||||
|
||||
Turn 1 asks the agent loop to call a tool, turn 2 ends the loop with plain text -
|
||||
exactly the two-step shape ``run_cowork`` exercises, with zero I/O.
|
||||
|
||||
It records every call it received (:attr:`FakeProvider.calls`) so a test can
|
||||
assert on what the layer above actually sent (message list, tool catalogue),
|
||||
which is how the characterization and contract suites pin current behaviour.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import itertools
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Dict, List, Optional, Sequence, Tuple
|
||||
|
||||
from cowork_local.providers.base import (
|
||||
CancelFn,
|
||||
Provider,
|
||||
ProviderError,
|
||||
TextCallback,
|
||||
ToolSpec,
|
||||
)
|
||||
|
||||
# One scripted tool call: (name, arguments). Ids are generated by the provider so
|
||||
# a test never has to invent them, mirroring what a real gateway does.
|
||||
ToolCallScript = Tuple[str, Dict[str, Any]]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ScriptedTurn:
|
||||
"""What :class:`FakeProvider` should do for ONE ``chat()`` call.
|
||||
|
||||
``text`` is streamed through ``on_text`` and returned as the assistant
|
||||
message content. ``reasoning`` goes to ``on_reasoning`` only - it must never
|
||||
leak into the answer, and asserting that is one of this double's jobs.
|
||||
|
||||
``tool_calls`` makes the agent loop run tools and come back for another turn;
|
||||
an empty tuple ends the loop.
|
||||
|
||||
``error``, when set, raises :class:`ProviderError` instead of answering, so
|
||||
error/recovery paths are testable without simulating a network fault.
|
||||
|
||||
``chunk_size`` > 0 splits ``text`` into fixed-size pieces to exercise
|
||||
chunk-boundary handling in stream consumers (the ``<think>`` splitter and the
|
||||
UI's incremental markdown renderer both have boundary logic worth covering).
|
||||
"""
|
||||
|
||||
text: str = ""
|
||||
reasoning: str = ""
|
||||
tool_calls: Sequence[ToolCallScript] = ()
|
||||
error: Optional[str] = None
|
||||
chunk_size: int = 0
|
||||
|
||||
|
||||
@dataclass
|
||||
class RecordedCall:
|
||||
"""A snapshot of one ``chat()`` invocation, for assertions after the fact."""
|
||||
|
||||
messages: List[Dict[str, Any]]
|
||||
tool_names: List[str]
|
||||
cancelled: bool = False
|
||||
|
||||
|
||||
class FakeProvider(Provider):
|
||||
"""A ``Provider`` that replays :class:`ScriptedTurn` objects.
|
||||
|
||||
Args:
|
||||
turns: the scripted turns, consumed in order.
|
||||
model: the model id reported through ``describe()`` / usage records.
|
||||
models: what :meth:`list_models` returns (Settings' "Load models").
|
||||
strict: when True (default) running past the end of the script raises
|
||||
``AssertionError``. That is intentional noise: a silent extra turn
|
||||
usually means the code under test looped more than the test author
|
||||
expected, and hiding it behind an empty answer would turn a real
|
||||
behaviour change into a passing test.
|
||||
"""
|
||||
|
||||
name = "fake"
|
||||
# The double can accept image content blocks, so vision code paths are
|
||||
# reachable in tests without a real vision-capable gateway.
|
||||
supports_vision = True
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
turns: Optional[Sequence[ScriptedTurn]] = None,
|
||||
*,
|
||||
model: str = "fake-model",
|
||||
models: Optional[Sequence[str]] = None,
|
||||
strict: bool = True,
|
||||
conf: Optional[Dict[str, Any]] = None,
|
||||
) -> None:
|
||||
super().__init__(dict(conf or {}, model=model))
|
||||
self._turns: List[ScriptedTurn] = list(turns or [])
|
||||
self._models = list(models or [model])
|
||||
self._strict = strict
|
||||
self._ids = itertools.count(1) # deterministic tool-call ids: call_1, call_2, ...
|
||||
self.calls: List[RecordedCall] = []
|
||||
|
||||
# -- introspection helpers used by tests ---------------------------- #
|
||||
@property
|
||||
def call_count(self) -> int:
|
||||
"""How many times the layer above asked this provider to run a turn."""
|
||||
return len(self.calls)
|
||||
|
||||
@property
|
||||
def remaining_turns(self) -> int:
|
||||
"""Scripted turns not consumed yet - assert 0 to prove the script was
|
||||
fully used (an unused turn means the code stopped earlier than intended)."""
|
||||
return len(self._turns)
|
||||
|
||||
def last_messages(self) -> List[Dict[str, Any]]:
|
||||
"""The message list sent on the most recent call (empty if never called)."""
|
||||
return self.calls[-1].messages if self.calls else []
|
||||
|
||||
# -- Provider contract ---------------------------------------------- #
|
||||
def chat(
|
||||
self,
|
||||
messages: List[Dict[str, Any]],
|
||||
tools: Optional[List[ToolSpec]] = None,
|
||||
on_text: Optional[TextCallback] = None,
|
||||
cancel: Optional[CancelFn] = None,
|
||||
on_reasoning: Optional[TextCallback] = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""Replay the next scripted turn, honouring cancel and both callbacks.
|
||||
|
||||
The message list is deep-ish copied into the recording because the agent
|
||||
loop keeps appending to the SAME list object; without the copy every
|
||||
recorded call would show the final state and assertions on "what was
|
||||
sent at step 1" would be meaningless.
|
||||
"""
|
||||
record = RecordedCall(
|
||||
messages=[dict(m) for m in messages],
|
||||
tool_names=[t.name for t in (tools or [])],
|
||||
)
|
||||
self.calls.append(record)
|
||||
|
||||
turn = self._next_turn()
|
||||
|
||||
# Checked before streaming anything: a provider that already knows the
|
||||
# caller gave up must not spend callbacks on text nobody will render.
|
||||
if self._is_cancelled(cancel):
|
||||
record.cancelled = True
|
||||
return {"role": "assistant", "content": "", "tool_calls": []}
|
||||
|
||||
if turn.error:
|
||||
raise ProviderError(turn.error)
|
||||
|
||||
if turn.reasoning and on_reasoning:
|
||||
on_reasoning(turn.reasoning)
|
||||
|
||||
for piece in self._stream_pieces(turn):
|
||||
# Re-checked between chunks so a mid-stream Stop truncates the answer
|
||||
# the same way a real streamed response does.
|
||||
if self._is_cancelled(cancel):
|
||||
record.cancelled = True
|
||||
break
|
||||
if on_text:
|
||||
on_text(piece)
|
||||
|
||||
return {
|
||||
"role": "assistant",
|
||||
"content": turn.text,
|
||||
"tool_calls": [
|
||||
{"id": f"call_{next(self._ids)}", "name": name, "arguments": dict(args)}
|
||||
for name, args in turn.tool_calls
|
||||
],
|
||||
}
|
||||
|
||||
def list_models(self) -> List[str]:
|
||||
"""Configured model ids. Clears ``last_error`` so ``test_connection()``
|
||||
reports success, matching how a healthy real provider behaves."""
|
||||
self.last_error = ""
|
||||
return list(self._models)
|
||||
|
||||
# -- internals ------------------------------------------------------- #
|
||||
def _next_turn(self) -> ScriptedTurn:
|
||||
"""Pop the next scripted turn, or fail loudly when the script ran out."""
|
||||
if self._turns:
|
||||
return self._turns.pop(0)
|
||||
if self._strict:
|
||||
raise AssertionError(
|
||||
f"FakeProvider script exhausted: chat() was called {len(self.calls)} "
|
||||
"time(s) but fewer turns were scripted. Add a ScriptedTurn, or pass "
|
||||
"strict=False if the extra call is genuinely expected."
|
||||
)
|
||||
return ScriptedTurn()
|
||||
|
||||
@staticmethod
|
||||
def _stream_pieces(turn: ScriptedTurn) -> List[str]:
|
||||
"""Split a turn's answer into the fragments to stream.
|
||||
|
||||
``chunk_size == 0`` streams the whole answer in one piece (the common
|
||||
case); a positive size slices it so tests can drive chunk-boundary logic.
|
||||
"""
|
||||
if not turn.text:
|
||||
return []
|
||||
if turn.chunk_size <= 0:
|
||||
return [turn.text]
|
||||
size = turn.chunk_size
|
||||
return [turn.text[i:i + size] for i in range(0, len(turn.text), size)]
|
||||
|
||||
|
||||
__all__ = ["FakeProvider", "ScriptedTurn", "RecordedCall"]
|
||||
@@ -1,99 +0,0 @@
|
||||
"""FakeToolExecutor - offline stand-in for the extra-tool executor (R01-T02).
|
||||
|
||||
``core.chat_agent.run_cowork`` routes any tool call whose name appears in
|
||||
``extra_tools`` to ``extra_executor(name, args)`` and expects back::
|
||||
|
||||
{"ok": bool, "output": str}
|
||||
|
||||
In production that callable reaches MCP servers, Microsoft 365 connectors and
|
||||
subprocesses. This double answers from a table instead, so the agent loop's tool
|
||||
branch is testable with no processes, no sockets and no credentials - and every
|
||||
invocation is recorded for assertions about what the agent actually asked for.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Callable, Dict, List, Optional, Union
|
||||
|
||||
from cowork_local.providers.base import ToolSpec
|
||||
|
||||
# A scripted answer is either the literal result dict, or a callable computing it
|
||||
# from the arguments (for tools whose output must depend on the input).
|
||||
ToolResult = Dict[str, Any]
|
||||
ScriptedResult = Union[ToolResult, Callable[[Dict[str, Any]], ToolResult]]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ToolInvocation:
|
||||
"""One recorded ``extra_executor(name, args)`` call."""
|
||||
|
||||
name: str
|
||||
args: Dict[str, Any]
|
||||
|
||||
|
||||
@dataclass
|
||||
class FakeToolExecutor:
|
||||
"""Callable test double for ``run_cowork(extra_executor=...)``.
|
||||
|
||||
Args:
|
||||
results: tool name -> scripted result (dict, or callable taking args).
|
||||
default: what to answer for a tool with no scripted result. ``None``
|
||||
(the default) answers with ``ok=False`` and an explicit message
|
||||
rather than raising - the production executor also reports unknown
|
||||
tools as a failed tool result, and matching that keeps the agent
|
||||
loop on its real code path instead of an exception path it would
|
||||
never take in production.
|
||||
"""
|
||||
|
||||
results: Dict[str, ScriptedResult] = field(default_factory=dict)
|
||||
default: Optional[ScriptedResult] = None
|
||||
calls: List[ToolInvocation] = field(default_factory=list)
|
||||
|
||||
def __call__(self, name: str, args: Dict[str, Any]) -> ToolResult:
|
||||
"""Record the invocation and return its scripted result."""
|
||||
self.calls.append(ToolInvocation(name=name, args=dict(args or {})))
|
||||
scripted = self.results.get(name, self.default)
|
||||
if scripted is None:
|
||||
return {"ok": False, "output": f"No fake result scripted for tool '{name}'."}
|
||||
# A callable lets one entry serve many different arguments (e.g. echo the
|
||||
# path it was asked to read) without scripting every combination.
|
||||
resolved = scripted(dict(args or {})) if callable(scripted) else dict(scripted)
|
||||
resolved.setdefault("ok", True)
|
||||
resolved.setdefault("output", "")
|
||||
return resolved
|
||||
|
||||
# -- introspection helpers used by tests ---------------------------- #
|
||||
@property
|
||||
def call_names(self) -> List[str]:
|
||||
"""Tool names in call order - the usual thing a test asserts on."""
|
||||
return [c.name for c in self.calls]
|
||||
|
||||
def called(self, name: str) -> bool:
|
||||
"""True when ``name`` was invoked at least once."""
|
||||
return any(c.name == name for c in self.calls)
|
||||
|
||||
def args_for(self, name: str) -> List[Dict[str, Any]]:
|
||||
"""Every argument dict this tool was called with, in order."""
|
||||
return [c.args for c in self.calls if c.name == name]
|
||||
|
||||
def specs(self) -> List[ToolSpec]:
|
||||
"""``ToolSpec`` entries for the scripted tools, ready to pass as
|
||||
``run_cowork(extra_tools=...)``.
|
||||
|
||||
The agent loop dispatches to ``extra_executor`` only for names present in
|
||||
``extra_tools``; generating the specs from the same table removes the
|
||||
chance of a test scripting a result the loop can never reach.
|
||||
"""
|
||||
return [
|
||||
ToolSpec(
|
||||
name=name,
|
||||
description=f"Fake tool '{name}' (test double).",
|
||||
# Permissive schema on purpose: these specs exist to register the
|
||||
# name with the agent loop, not to validate arguments.
|
||||
parameters={"type": "object", "properties": {}, "additionalProperties": True},
|
||||
)
|
||||
for name in self.results
|
||||
]
|
||||
|
||||
|
||||
__all__ = ["FakeToolExecutor", "ToolInvocation"]
|
||||
@@ -0,0 +1,44 @@
|
||||
"""ToolPolicyGateway giả — để N3 (Co4E) chạy được khi Team Hoa chưa cài đặt.
|
||||
|
||||
Mặc định cho qua hết, vì phần lớn test Co4E quan tâm tới luồng workflow chứ
|
||||
không phải chính sách. Test nào cần kiểm nhánh bị chặn thì lập trình câu trả
|
||||
lời::
|
||||
|
||||
gate = FakeToolPolicyGateway(rules={"run_command": deny("cấm trong Co4E")})
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Callable, Dict
|
||||
|
||||
from cowork_local.domain.security.tool_policy import (
|
||||
PolicyDecision, ToolCallRequest, allow,
|
||||
)
|
||||
|
||||
|
||||
class FakeToolPolicyGateway:
|
||||
"""Cổng chính sách trong bộ nhớ, có ghi lại đã hỏi những gì."""
|
||||
|
||||
def __init__(self, rules: Dict[str, PolicyDecision] | None = None,
|
||||
default: PolicyDecision | None = None,
|
||||
decide: Callable[[ToolCallRequest], PolicyDecision] | None = None):
|
||||
#: {tên tool: quyết định} — tra trước default
|
||||
self.rules = dict(rules or {})
|
||||
self.default = default or allow()
|
||||
#: hàm tự quyết, dùng khi cần logic phức tạp hơn tra bảng
|
||||
self._decide = decide
|
||||
#: mọi lời gọi đã đi qua — để test khẳng định "có hỏi cổng không"
|
||||
self.seen: list[ToolCallRequest] = []
|
||||
|
||||
def check(self, request: ToolCallRequest) -> PolicyDecision:
|
||||
self.seen.append(request)
|
||||
if self._decide is not None:
|
||||
return self._decide(request)
|
||||
return self.rules.get(request.name, self.default)
|
||||
|
||||
# ---- tiện cho test --------------------------------------------------
|
||||
def asked_for(self, name: str) -> bool:
|
||||
return any(r.name == name for r in self.seen)
|
||||
|
||||
@property
|
||||
def call_count(self) -> int:
|
||||
return len(self.seen)
|
||||
@@ -1,8 +0,0 @@
|
||||
"""Integration tests: real widgets, real services, no network (R10-T01 layout).
|
||||
|
||||
These build actual Qt widgets offscreen (``QT_QPA_PLATFORM=offscreen``) and run
|
||||
a turn end to end with a scripted :class:`FakeProvider`. They are slower than
|
||||
the unit suite - a QApplication has to exist - and are what proves the seams
|
||||
introduced by R03/R04 are actually wired into the screens, not just correct in
|
||||
isolation.
|
||||
"""
|
||||
@@ -1,201 +0,0 @@
|
||||
"""End-to-end check that the Cowork screen really runs turns through the
|
||||
application layer (R04-T04).
|
||||
|
||||
The unit tests prove ``ConversationApplicationService`` behaves correctly; this
|
||||
one proves ``ui/cowork_tab.py::build_job`` actually goes through it, on a real
|
||||
(offscreen) widget, with a scripted provider instead of a network call.
|
||||
|
||||
It also pins the property that motivated R04-T01: the turn runs on the state
|
||||
captured at SUBMIT time, so a user editing the conversation while a turn is in
|
||||
flight cannot change what that turn sends.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List
|
||||
|
||||
import pytest
|
||||
|
||||
os.environ.setdefault("QT_QPA_PLATFORM", "offscreen")
|
||||
|
||||
from cowork_local.config import AppConfig # noqa: E402
|
||||
from cowork_local.core import chat_agent # noqa: E402
|
||||
from cowork_local.state import AppContext # noqa: E402
|
||||
from tests.fakes import FakeProvider, ScriptedTurn # noqa: E402
|
||||
|
||||
pytest.importorskip("PySide6", reason="Qt is required for the integration suite")
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def qt_app():
|
||||
"""One QApplication for the module - Qt allows only a single instance."""
|
||||
from PySide6.QtWidgets import QApplication
|
||||
|
||||
return QApplication.instance() or QApplication([])
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def cowork_tab(qt_app, tmp_path: Path, monkeypatch):
|
||||
"""A real CoworkTab on a throwaway config, with ambient inputs neutralised."""
|
||||
monkeypatch.setattr(chat_agent, "active_skills_text", lambda: "")
|
||||
monkeypatch.setattr(chat_agent, "load_rules", lambda: "")
|
||||
from cowork_local.core import audit_log
|
||||
|
||||
monkeypatch.setattr(audit_log, "AUDIT_DIR", tmp_path / "audit")
|
||||
|
||||
from cowork_local.ui.cowork_tab import CoworkTab
|
||||
|
||||
ctx = AppContext(AppConfig.load(tmp_path / "config.json"))
|
||||
# Agent Security's prompt validation is ON by default and spends an EXTRA
|
||||
# provider call reviewing the request before the agent loop starts (see
|
||||
# core/agent_security.py::enforce_prompt). That is real behaviour - pinned
|
||||
# by its own test below - but it would make every other test here script a
|
||||
# turn that has nothing to do with what it is checking.
|
||||
ctx.config.agent_security["enabled"] = False
|
||||
return CoworkTab(ctx)
|
||||
|
||||
|
||||
class _StubWorker:
|
||||
"""The slice of ``core.worker.AgentWorker`` a job actually touches."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.events: List[Dict[str, Any]] = []
|
||||
self.gates_requested = 0
|
||||
self._cancelled = False
|
||||
|
||||
def emit_event(self, payload: Dict[str, Any]) -> None:
|
||||
self.events.append(payload)
|
||||
|
||||
def is_cancelled(self) -> bool:
|
||||
return self._cancelled
|
||||
|
||||
def new_gate(self, _mode: str, **_kwargs) -> Any:
|
||||
self.gates_requested += 1
|
||||
return None
|
||||
|
||||
def cancel(self) -> None:
|
||||
self._cancelled = True
|
||||
|
||||
|
||||
def _run_job(tab, worker, provider, text="hello", messages=None, out_dir=None):
|
||||
"""Build the tab's job with ``provider`` pinned, then run it like the worker
|
||||
thread would."""
|
||||
tab.build_provider = lambda: provider # what routing/agent selection resolves to
|
||||
job = tab.build_job(text, messages if messages is not None
|
||||
else [{"role": "user", "content": text}], out_dir)
|
||||
return job(worker)
|
||||
|
||||
|
||||
def test_a_turn_runs_through_the_service_and_returns_history(cowork_tab, tmp_path):
|
||||
provider = FakeProvider([ScriptedTurn(text="Hello from the fake.")])
|
||||
worker = _StubWorker()
|
||||
|
||||
result = _run_job(cowork_tab, worker, provider, out_dir=tmp_path / "turn")
|
||||
|
||||
assert provider.call_count == 1
|
||||
# Same return contract as before the refactor - _cleanup_turn reads both keys.
|
||||
assert set(result) == {"messages", "turn_dir"}
|
||||
assert [m["role"] for m in result["messages"]] == ["system", "user", "assistant"]
|
||||
assert result["messages"][-1]["content"] == "Hello from the fake."
|
||||
|
||||
|
||||
def test_the_widget_still_receives_the_legacy_event_dicts(cowork_tab, tmp_path):
|
||||
"""The chat widgets consume dicts and are not migrated until EPIC R08, so
|
||||
the typed events must render back into exactly what they already handle -
|
||||
plus the new end-of-turn signal, which the if/elif dispatch ignores."""
|
||||
provider = FakeProvider([ScriptedTurn(text="Hi")])
|
||||
worker = _StubWorker()
|
||||
|
||||
_run_job(cowork_tab, worker, provider, out_dir=tmp_path / "turn")
|
||||
|
||||
assert [e["type"] for e in worker.events] == ["text", "assistant_done", "turn_completed"]
|
||||
assert worker.events[0] == {"type": "text", "delta": "Hi"}
|
||||
|
||||
|
||||
def test_the_turn_ignores_messages_added_after_it_was_submitted(cowork_tab, tmp_path):
|
||||
"""The bug ConversationExecutionRequest exists to prevent: the panel keeps
|
||||
appending to its own list while a turn is in flight."""
|
||||
provider = FakeProvider([ScriptedTurn(text="ok")])
|
||||
worker = _StubWorker()
|
||||
live_messages = [{"role": "user", "content": "first question"}]
|
||||
|
||||
job_result = _run_job(cowork_tab, worker, provider,
|
||||
messages=live_messages, out_dir=tmp_path / "turn")
|
||||
|
||||
# Simulate the user typing a second message DURING the turn by mutating the
|
||||
# list the panel handed over. The already-sent conversation must not include it.
|
||||
live_messages.append({"role": "user", "content": "typed while running"})
|
||||
|
||||
sent = provider.calls[0].messages
|
||||
assert [m["content"] for m in sent if m["role"] == "user"] == ["first question"]
|
||||
assert "typed while running" not in str(job_result["messages"])
|
||||
|
||||
|
||||
def test_a_failing_turn_still_raises_so_the_worker_reports_it(cowork_tab, tmp_path):
|
||||
"""core/worker.py turns an exception into the `failed` signal the chat panel
|
||||
already handles; swallowing it here would show a successful turn with no
|
||||
answer instead of an error."""
|
||||
provider = FakeProvider([ScriptedTurn(error="gateway down"),
|
||||
ScriptedTurn(error="gateway down")])
|
||||
worker = _StubWorker()
|
||||
|
||||
with pytest.raises(Exception) as excinfo:
|
||||
_run_job(cowork_tab, worker, provider, out_dir=tmp_path / "turn")
|
||||
|
||||
assert "gateway down" in str(excinfo.value)
|
||||
# The error was still reported as an event before being re-raised.
|
||||
assert any(e["type"] == "error" for e in worker.events)
|
||||
|
||||
|
||||
def test_a_permission_gate_is_only_requested_when_the_workspace_asks_for_it(
|
||||
cowork_tab, tmp_path, monkeypatch):
|
||||
provider = FakeProvider([ScriptedTurn(text="ok"), ScriptedTurn(text="ok")])
|
||||
worker = _StubWorker()
|
||||
|
||||
monkeypatch.setattr(cowork_tab.ctx, "project_confirm_commands", lambda: False)
|
||||
_run_job(cowork_tab, worker, provider, out_dir=tmp_path / "a")
|
||||
assert worker.gates_requested == 0
|
||||
|
||||
monkeypatch.setattr(cowork_tab.ctx, "project_confirm_commands", lambda: True)
|
||||
_run_job(cowork_tab, worker, provider, out_dir=tmp_path / "b")
|
||||
assert worker.gates_requested == 1
|
||||
|
||||
|
||||
def test_a_tool_turn_writes_into_this_turns_own_output_folder(cowork_tab, tmp_path):
|
||||
"""Turn isolation: each turn writes into its own directory so parallel turns
|
||||
cannot clobber each other's files."""
|
||||
provider = FakeProvider([
|
||||
ScriptedTurn(tool_calls=[("save_file", {"filename": "n.md", "content": "x"})]),
|
||||
ScriptedTurn(text="Saved."),
|
||||
])
|
||||
worker = _StubWorker()
|
||||
turn_dir = tmp_path / "turn-1"
|
||||
|
||||
result = _run_job(cowork_tab, worker, provider, out_dir=turn_dir)
|
||||
|
||||
assert result["turn_dir"] == str(turn_dir)
|
||||
assert [p.name for p in turn_dir.iterdir()] and turn_dir.exists()
|
||||
assert any(e["type"] == "tool_result" and e["ok"] for e in worker.events)
|
||||
|
||||
|
||||
def test_agent_security_still_reviews_the_request_before_the_turn_runs(
|
||||
cowork_tab, tmp_path):
|
||||
"""Characterisation, not a new behaviour: with Agent Security enabled (the
|
||||
shipped default) a turn costs an EXTRA provider call, because the request is
|
||||
reviewed against the rulebase before the agent loop starts.
|
||||
|
||||
Pinned here because it is invisible from the call site and easy to break -
|
||||
routing a turn through the application layer must not skip the review.
|
||||
"""
|
||||
cowork_tab.ctx.config.agent_security["enabled"] = True
|
||||
provider = FakeProvider([
|
||||
ScriptedTurn(text="ALLOW"), # the security pre-flight review
|
||||
ScriptedTurn(text="the answer"), # the turn itself
|
||||
])
|
||||
worker = _StubWorker()
|
||||
|
||||
result = _run_job(cowork_tab, worker, provider, out_dir=tmp_path / "turn")
|
||||
|
||||
assert provider.call_count == 2
|
||||
assert result["messages"][-1]["content"] == "the answer"
|
||||
@@ -1,256 +0,0 @@
|
||||
"""The three chat surfaces really route through the shared service (R03-T04/T05).
|
||||
|
||||
The unit suite proves ``RoutingApplicationService`` decides correctly against a
|
||||
fake router. This file proves the three widgets that used to own a private copy
|
||||
of that algorithm now call it, on real (offscreen) widgets:
|
||||
|
||||
* ``ui/chat_panel.py::_apply_routing`` (Cowork)
|
||||
* ``ui/co4e_tab.py::_apply_co4e_routing`` (Co4E)
|
||||
* ``ui/folder_tab.py::_ai_apply_routing`` (AI-Edit)
|
||||
|
||||
It also pins the Manual-mode handshake, including the field contract the
|
||||
existing confirm dialog reads off the decision - the one place where the new
|
||||
``RoutingDecision`` has to look like the legacy ``SwitchDecision`` it replaced.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from pathlib import Path
|
||||
from typing import Any, List, Optional, Tuple
|
||||
|
||||
import pytest
|
||||
|
||||
os.environ.setdefault("QT_QPA_PLATFORM", "offscreen")
|
||||
|
||||
from cowork_local.application.model_routing import ( # noqa: E402
|
||||
RoutingApplicationService,
|
||||
RoutingDecision,
|
||||
RoutingMode,
|
||||
)
|
||||
from cowork_local.config import AppConfig # noqa: E402
|
||||
from cowork_local.state import AppContext # noqa: E402
|
||||
|
||||
pytest.importorskip("PySide6", reason="Qt is required for the integration suite")
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def qt_app():
|
||||
from PySide6.QtWidgets import QApplication
|
||||
|
||||
return QApplication.instance() or QApplication([])
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def ctx(qt_app, tmp_path: Path) -> AppContext:
|
||||
return AppContext(AppConfig.load(tmp_path / "config.json"))
|
||||
|
||||
|
||||
class _FakeRouteResult:
|
||||
"""Shaped like ``core.routing.service.RouteResult``."""
|
||||
|
||||
def __init__(self, provider: str, model: str, gain: float = 0.4,
|
||||
task: str = "coding") -> None:
|
||||
self.should_switch = True
|
||||
self._target = (provider, model)
|
||||
self.task_type = type("_T", (), {"value": task})()
|
||||
self.decision = type("_D", (), {"score_gain": gain, "reason": "better fit"})()
|
||||
|
||||
def target(self) -> Optional[Tuple[str, str]]:
|
||||
return self._target
|
||||
|
||||
|
||||
class _FakeRouter:
|
||||
"""Minimal RoutingPort: always proposes the same switch, records the surface."""
|
||||
|
||||
def __init__(self, provider="anthropic", model="claude-sonnet-4-6") -> None:
|
||||
self.result = _FakeRouteResult(provider, model)
|
||||
self.surfaces: List[str] = []
|
||||
|
||||
def route(self, surface, prompt, current_provider, current_model, **kwargs):
|
||||
self.surfaces.append(surface)
|
||||
return self.result
|
||||
|
||||
|
||||
def _install(ctx: AppContext, mode: str) -> _FakeRouter:
|
||||
"""Wire a fake router into the context and force ``mode`` on every surface."""
|
||||
router = _FakeRouter()
|
||||
service = RoutingApplicationService(router, mode_reader=lambda _surface: mode)
|
||||
ctx._routing_application = service # already-built instance; accessor returns it
|
||||
return router
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Cowork chat
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_cowork_applies_an_auto_switch_to_the_next_turn(ctx):
|
||||
from cowork_local.ui.cowork_tab import CoworkTab
|
||||
|
||||
router = _install(ctx, "auto")
|
||||
tab = CoworkTab(ctx)
|
||||
turn: dict = {"bubbles": []}
|
||||
|
||||
tab._apply_routing("write a function", turn)
|
||||
|
||||
assert router.surfaces == [tab.kind]
|
||||
# build_provider() honours these for THIS turn only.
|
||||
assert (tab._routed_provider, tab._routed_model) == ("anthropic", "claude-sonnet-4-6")
|
||||
assert turn["bubbles"], "the user must be told the model was switched"
|
||||
|
||||
|
||||
def test_cowork_leaves_the_model_alone_when_routing_is_off(ctx):
|
||||
from cowork_local.ui.cowork_tab import CoworkTab
|
||||
|
||||
router = _install(ctx, "off")
|
||||
tab = CoworkTab(ctx)
|
||||
turn: dict = {"bubbles": []}
|
||||
|
||||
tab._apply_routing("write a function", turn)
|
||||
|
||||
assert router.surfaces == []
|
||||
assert (tab._routed_provider, tab._routed_model) == (None, None)
|
||||
assert turn["bubbles"] == []
|
||||
|
||||
|
||||
def test_cowork_manual_mode_switches_only_after_the_dialog_approves(ctx, monkeypatch):
|
||||
from cowork_local.ui import chat_panel as chat_panel_module
|
||||
from cowork_local.ui.cowork_tab import CoworkTab
|
||||
|
||||
_install(ctx, "manual")
|
||||
tab = CoworkTab(ctx)
|
||||
asked: List[Any] = []
|
||||
monkeypatch.setattr(tab, "_confirm_routing_switch",
|
||||
lambda decision: asked.append(decision) or True)
|
||||
turn: dict = {"bubbles": []}
|
||||
|
||||
tab._apply_routing("write a function", turn)
|
||||
|
||||
assert len(asked) == 1
|
||||
assert (tab._routed_provider, tab._routed_model) == ("anthropic", "claude-sonnet-4-6")
|
||||
|
||||
|
||||
def test_cowork_manual_mode_keeps_the_model_when_the_dialog_is_declined(ctx, monkeypatch):
|
||||
from cowork_local.ui.cowork_tab import CoworkTab
|
||||
|
||||
_install(ctx, "manual")
|
||||
tab = CoworkTab(ctx)
|
||||
monkeypatch.setattr(tab, "_confirm_routing_switch", lambda _decision: False)
|
||||
turn: dict = {"bubbles": []}
|
||||
|
||||
tab._apply_routing("write a function", turn)
|
||||
|
||||
assert (tab._routed_provider, tab._routed_model) == (None, None)
|
||||
assert turn["bubbles"] == []
|
||||
|
||||
|
||||
def test_a_pinned_admin_agent_still_wins_over_routing(ctx):
|
||||
"""An explicitly chosen Admin agent pins its own provider/model; routing must
|
||||
not override a deliberate user choice."""
|
||||
from cowork_local.ui.cowork_tab import CoworkTab
|
||||
|
||||
router = _install(ctx, "auto")
|
||||
tab = CoworkTab(ctx)
|
||||
tab._admin_agent = object()
|
||||
turn: dict = {"bubbles": []}
|
||||
|
||||
tab._apply_routing("write a function", turn)
|
||||
|
||||
assert router.surfaces == []
|
||||
assert (tab._routed_provider, tab._routed_model) == (None, None)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Co4E
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_co4e_routes_on_its_own_surface_key_and_returns_the_model(ctx):
|
||||
from cowork_local.ui.co4e_tab import Co4ETab
|
||||
|
||||
router = _install(ctx, "auto")
|
||||
tab = Co4ETab(ctx)
|
||||
|
||||
model = tab._apply_co4e_routing("build me a flow")
|
||||
|
||||
assert router.surfaces == ["co4e"]
|
||||
assert model == "claude-sonnet-4-6"
|
||||
assert tab._co4e_routed_provider == "anthropic"
|
||||
|
||||
|
||||
def test_co4e_returns_an_empty_model_when_routing_is_off(ctx):
|
||||
"""'' means "use the provider default" - the contract _run_chat_turn expects."""
|
||||
from cowork_local.ui.co4e_tab import Co4ETab
|
||||
|
||||
_install(ctx, "off")
|
||||
tab = Co4ETab(ctx)
|
||||
|
||||
assert tab._apply_co4e_routing("build me a flow") == ""
|
||||
assert tab._co4e_routed_provider is None
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# AI-Edit
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_ai_edit_routes_on_its_own_surface_key(ctx):
|
||||
from cowork_local.ui.folder_tab import FolderTab
|
||||
|
||||
router = _install(ctx, "auto")
|
||||
tab = FolderTab(ctx)
|
||||
|
||||
tab._ai_apply_routing("rename this variable")
|
||||
|
||||
assert router.surfaces == ["ai_edit"]
|
||||
assert (tab._ai_routed_provider, tab._ai_routed_model) == (
|
||||
"anthropic", "claude-sonnet-4-6")
|
||||
|
||||
|
||||
def test_ai_edit_pins_the_coding_task_type(ctx):
|
||||
"""An edit instruction is never a QA question, so AI-Edit skips
|
||||
classification entirely - the constraint has to survive the move into the
|
||||
shared service or it is silently dropped."""
|
||||
from cowork_local.core.routing.models import TaskType
|
||||
from cowork_local.ui.folder_tab import FolderTab
|
||||
|
||||
seen: List[Any] = []
|
||||
|
||||
class _Recorder(_FakeRouter):
|
||||
def route(self, surface, prompt, current_provider, current_model, **kwargs):
|
||||
seen.append(kwargs.get("task_type"))
|
||||
return super().route(surface, prompt, current_provider, current_model, **kwargs)
|
||||
|
||||
ctx._routing_application = RoutingApplicationService(
|
||||
_Recorder(), mode_reader=lambda _s: "auto")
|
||||
tab = FolderTab(ctx)
|
||||
|
||||
tab._ai_apply_routing("rename this variable")
|
||||
|
||||
assert seen == [TaskType.CODING]
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# The confirm dialog's field contract
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_the_decision_exposes_exactly_what_the_confirm_dialog_reads():
|
||||
"""``ui/routing_toggle.py::confirm_switch`` is not migrated until EPIC R08,
|
||||
so it still reads ``from_model``/``to_model`` as ``provider/model`` candidate
|
||||
keys and splits them. A rename here would blow up inside a modal dialog -
|
||||
the one place a failure is hardest to see in a test run."""
|
||||
from cowork_local.core.routing.models import split_key
|
||||
|
||||
decision = RoutingDecision(
|
||||
mode=RoutingMode.MANUAL, provider="anthropic", model="claude-sonnet-4-6",
|
||||
switched=True, task_type="coding", score_gain=0.31, reason="better fit",
|
||||
previous_provider="openai_compat", previous_model="gpt-4o-mini",
|
||||
)
|
||||
|
||||
assert split_key(decision.from_model)[1] == "gpt-4o-mini"
|
||||
assert split_key(decision.to_model)[1] == "claude-sonnet-4-6"
|
||||
assert decision.task_type == "coding"
|
||||
assert f"{decision.score_gain:.2f}" == "0.31"
|
||||
assert decision.reason == "better fit"
|
||||
|
||||
|
||||
def test_a_first_turn_with_no_current_model_yields_an_empty_from_model():
|
||||
"""split_key() is only called when from_model is truthy, so an unset current
|
||||
model must produce "" rather than a bare "provider/"."""
|
||||
decision = RoutingDecision(mode=RoutingMode.AUTO, provider="anthropic",
|
||||
model="claude", switched=True)
|
||||
|
||||
assert decision.from_model == ""
|
||||
@@ -1,178 +0,0 @@
|
||||
"""End-to-end check of the Schedule Task path after R04-T05.
|
||||
|
||||
``core/task_executors.py::_run_agent`` used to assemble its own ``run_cowork``
|
||||
call, in parallel with ``ui/cowork_tab.py`` doing the same thing slightly
|
||||
differently. It now goes through ``ConversationApplicationService``, and the
|
||||
things most at risk from that change are exactly what this file pins:
|
||||
|
||||
* the unattended run still returns the answer text the scheduler writes to output.md
|
||||
* History is still re-saved from the LIVE message list after every assistant
|
||||
message, so a long run shows progress when reopened mid-flight
|
||||
* ``update_plan`` tracking still works, so a task whose checklist is unfinished
|
||||
is not reported as done
|
||||
* a failed run still raises, because ``execute_task`` writes error.txt from it
|
||||
|
||||
No Qt and no network: the provider is scripted and History is redirected into a
|
||||
tmp folder.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List
|
||||
|
||||
import pytest
|
||||
|
||||
from cowork_local.config import AppConfig
|
||||
from cowork_local.core import audit_log, chat_agent, task_executors
|
||||
from cowork_local.state import AppContext
|
||||
from tests.fakes import FakeProvider, ScriptedTurn
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def task_ctx(tmp_path: Path, monkeypatch):
|
||||
"""An AppContext whose History and audit log live in a tmp folder."""
|
||||
monkeypatch.setattr(chat_agent, "active_skills_text", lambda: "")
|
||||
monkeypatch.setattr(chat_agent, "load_rules", lambda: "")
|
||||
monkeypatch.setattr(audit_log, "AUDIT_DIR", tmp_path / "audit")
|
||||
|
||||
ctx = AppContext(AppConfig.load(tmp_path / "config.json"))
|
||||
# Same reason as the Cowork integration suite: the security pre-flight costs
|
||||
# an extra provider call that has nothing to do with what is being tested.
|
||||
ctx.config.agent_security["enabled"] = False
|
||||
monkeypatch.setattr(ctx.config, "history_dir", lambda: tmp_path / "history")
|
||||
return ctx
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def history_saves(monkeypatch) -> List[List[Dict[str, Any]]]:
|
||||
"""Capture a SNAPSHOT of the messages at each History save.
|
||||
|
||||
Snapshotting matters: the engine keeps appending to the same list, so
|
||||
storing the list itself would make every recorded save look identical to the
|
||||
final state and the "live progress" assertion would prove nothing.
|
||||
"""
|
||||
saves: List[List[Dict[str, Any]]] = []
|
||||
|
||||
def fake_save(_dir, _kind, _session_id, messages, **_kwargs):
|
||||
saves.append([dict(m) for m in messages])
|
||||
|
||||
from cowork_local.core import history
|
||||
|
||||
monkeypatch.setattr(history, "save_conversation", fake_save)
|
||||
return saves
|
||||
|
||||
|
||||
def _run(ctx, provider, prompt="do the thing", out_dir: Path = None, **kwargs):
|
||||
"""Run one unattended cowork task with ``provider`` pinned."""
|
||||
ctx.build_active_provider = lambda: provider
|
||||
events: List[Dict[str, Any]] = []
|
||||
result = task_executors._run_agent(
|
||||
ctx, "cowork", prompt, out_dir, events.append, lambda: False,
|
||||
title=kwargs.pop("title", "T1"), **kwargs)
|
||||
return result, events
|
||||
|
||||
|
||||
def test_an_unattended_cowork_run_returns_the_answer(task_ctx, tmp_path, history_saves):
|
||||
provider = FakeProvider([ScriptedTurn(text="task answer")])
|
||||
|
||||
(answer, timed_out, incomplete), events = _run(task_ctx, provider,
|
||||
out_dir=tmp_path / "out")
|
||||
|
||||
assert answer == "task answer"
|
||||
assert timed_out is False
|
||||
assert incomplete == ""
|
||||
assert provider.call_count == 1
|
||||
|
||||
|
||||
def test_the_scheduler_still_gets_history_ready_before_the_turn_events(
|
||||
task_ctx, tmp_path, history_saves):
|
||||
"""The scheduler refreshes the History panel on this event, so a running
|
||||
task's conversation shows up while it runs."""
|
||||
provider = FakeProvider([ScriptedTurn(text="ok")])
|
||||
|
||||
_, events = _run(task_ctx, provider, out_dir=tmp_path / "out")
|
||||
|
||||
assert [e["type"] for e in events] == [
|
||||
"history_ready", "text", "assistant_done", "turn_completed"]
|
||||
|
||||
|
||||
def test_history_is_resaved_from_the_live_conversation_during_the_run(
|
||||
task_ctx, tmp_path, history_saves):
|
||||
"""The reason ``begin_turn()`` exists: the service builds its own message
|
||||
list, and the scheduler needs THAT list - not the pre-turn copy - or the
|
||||
mid-run saves would only ever contain the original user message.
|
||||
"""
|
||||
provider = FakeProvider([
|
||||
ScriptedTurn(tool_calls=[("save_file", {"filename": "a.md", "content": "x"})]),
|
||||
ScriptedTurn(text="Saved."),
|
||||
])
|
||||
|
||||
_run(task_ctx, provider, out_dir=tmp_path / "out")
|
||||
|
||||
# At least one save DURING the run already carried an assistant message,
|
||||
# and the final save carries the whole conversation.
|
||||
assert len(history_saves) >= 3 # initial + per assistant_done + final
|
||||
assert any(any(m["role"] == "assistant" for m in save)
|
||||
for save in history_saves[1:-1])
|
||||
assert [m["role"] for m in history_saves[-1]] == [
|
||||
"system", "user", "assistant", "tool", "assistant"]
|
||||
|
||||
|
||||
def test_an_unfinished_plan_is_reported_so_the_task_is_not_marked_done(
|
||||
task_ctx, tmp_path, history_saves):
|
||||
"""plan_set tracking runs through the same emit path; losing it would let a
|
||||
task whose own checklist says "not finished" be reported as successful."""
|
||||
provider = FakeProvider([
|
||||
ScriptedTurn(tool_calls=[("update_plan", {"steps": [
|
||||
{"title": "step one", "status": "running"}]})]),
|
||||
ScriptedTurn(text="stopping here"),
|
||||
])
|
||||
|
||||
(_answer, _timed_out, incomplete), _events = _run(task_ctx, provider,
|
||||
out_dir=tmp_path / "out")
|
||||
|
||||
assert incomplete != ""
|
||||
|
||||
|
||||
def test_a_completed_plan_reports_no_incompleteness(task_ctx, tmp_path, history_saves):
|
||||
provider = FakeProvider([
|
||||
ScriptedTurn(tool_calls=[("update_plan", {"steps": [
|
||||
{"title": "step one", "status": "done"}]})]),
|
||||
ScriptedTurn(text="all done"),
|
||||
])
|
||||
|
||||
(_answer, _timed_out, incomplete), _events = _run(task_ctx, provider,
|
||||
out_dir=tmp_path / "out")
|
||||
|
||||
assert incomplete == ""
|
||||
|
||||
|
||||
def test_a_failed_run_still_raises_so_execute_task_writes_error_txt(
|
||||
task_ctx, tmp_path, history_saves):
|
||||
provider = FakeProvider([ScriptedTurn(error="provider down"),
|
||||
ScriptedTurn(error="provider down")])
|
||||
|
||||
with pytest.raises(Exception) as excinfo:
|
||||
_run(task_ctx, provider, out_dir=tmp_path / "out")
|
||||
|
||||
assert "provider down" in str(excinfo.value)
|
||||
# The partial conversation is still saved - it is exactly what the user
|
||||
# needs to see after a failure.
|
||||
assert history_saves
|
||||
|
||||
|
||||
def test_a_per_task_provider_override_is_honoured(task_ctx, tmp_path, history_saves):
|
||||
"""A task can pin its own provider/model; the service must use that one, not
|
||||
the machine's Settings default."""
|
||||
default_provider = FakeProvider([], strict=True)
|
||||
task_provider = FakeProvider([ScriptedTurn(text="from the pinned model")])
|
||||
task_ctx.build_active_provider = lambda: default_provider
|
||||
task_ctx.build_provider_for = lambda _name, _model: task_provider
|
||||
|
||||
(answer, _timed_out, _incomplete) = task_executors._run_agent(
|
||||
task_ctx, "cowork", "go", tmp_path / "out", lambda _e: None, lambda: False,
|
||||
title="T", provider_name="anthropic", model="claude")[0:3]
|
||||
|
||||
assert answer == "from the pinned model"
|
||||
assert default_provider.call_count == 0
|
||||
assert task_provider.call_count == 1
|
||||
@@ -0,0 +1,105 @@
|
||||
"""AtomicJsonFile — R02-T01. Test tiêm lỗi, đúng như cột nghiệm thu của plan.md.
|
||||
|
||||
Cách kiểm: cắt ngang giữa lúc ghi rồi khẳng định file cũ **còn nguyên**. Nếu
|
||||
chỉ test "ghi rồi đọc lại thấy đúng" thì `path.write_text()` cũ cũng qua — mà
|
||||
đó chính là thứ ta đang thay.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from cowork_local.infrastructure.persistence.json.atomic_json_file import AtomicJsonFile
|
||||
|
||||
|
||||
def test_ghi_roi_doc_lai(tmp_path):
|
||||
f = AtomicJsonFile(tmp_path / "cau_hinh.json")
|
||||
f.write({"theme": "dark", "ngôn ngữ": "vi"})
|
||||
assert f.read() == {"theme": "dark", "ngôn ngữ": "vi"}
|
||||
|
||||
|
||||
def test_chua_co_file_thi_tra_mac_dinh(tmp_path):
|
||||
f = AtomicJsonFile(tmp_path / "chua-ton-tai.json")
|
||||
assert f.read(default={"theme": "dark"}) == {"theme": "dark"}
|
||||
assert f.exists() is False
|
||||
|
||||
|
||||
def test_chet_giua_luc_ghi_thi_file_cu_con_nguyen(tmp_path, monkeypatch):
|
||||
"""Lõi của R02-T01.
|
||||
|
||||
Giả lập mất điện đúng lúc: cho ``os.replace`` ném lỗi. Đây là bước cuối
|
||||
cùng, tức là dữ liệu mới đã nằm trong file tạm rồi — nếu cài đặt sai theo
|
||||
kiểu ghi đè thẳng, file đích lúc này đã hỏng.
|
||||
"""
|
||||
path = tmp_path / "cau_hinh.json"
|
||||
f = AtomicJsonFile(path)
|
||||
f.write({"phiên bản": 1, "quan trọng": "đừng mất"})
|
||||
|
||||
def no_dien(*args, **kwargs):
|
||||
raise OSError("mô phỏng mất điện")
|
||||
|
||||
monkeypatch.setattr(os, "replace", no_dien)
|
||||
with pytest.raises(OSError):
|
||||
f.write({"phiên bản": 2})
|
||||
|
||||
# bản cũ phải còn y nguyên
|
||||
assert f.read() == {"phiên bản": 1, "quan trọng": "đừng mất"}
|
||||
|
||||
|
||||
def test_khong_de_lai_rac_tmp_khi_ghi_hong(tmp_path, monkeypatch):
|
||||
path = tmp_path / "cau_hinh.json"
|
||||
f = AtomicJsonFile(path)
|
||||
f.write({"a": 1})
|
||||
|
||||
monkeypatch.setattr(os, "replace", lambda *a, **k: (_ for _ in ()).throw(OSError("x")))
|
||||
with pytest.raises(OSError):
|
||||
f.write({"a": 2})
|
||||
|
||||
con_lai = [p.name for p in tmp_path.iterdir()]
|
||||
assert con_lai == ["cau_hinh.json"], f"còn rác: {con_lai}"
|
||||
|
||||
|
||||
def test_file_hong_thi_cach_ly_va_tra_mac_dinh(tmp_path):
|
||||
"""Hỏng cấu hình không được chặn khởi động — giữ đúng hành vi config.py
|
||||
hiện tại, nhưng thêm phần giữ lại bản hỏng để còn cứu."""
|
||||
path = tmp_path / "cau_hinh.json"
|
||||
path.write_text("{ đây không phải json", encoding="utf-8")
|
||||
f = AtomicJsonFile(path)
|
||||
|
||||
assert f.read(default={"theme": "dark"}) == {"theme": "dark"}
|
||||
assert not path.exists(), "file hỏng phải được dời đi"
|
||||
bad = list(tmp_path.glob("*.bad-*"))
|
||||
assert len(bad) == 1, "phải giữ lại bản hỏng để cứu tay"
|
||||
assert "đây không phải json" in bad[0].read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def test_ghi_de_nhieu_lan_van_dung(tmp_path):
|
||||
f = AtomicJsonFile(tmp_path / "dem.json")
|
||||
for i in range(20):
|
||||
f.write({"lần": i})
|
||||
assert f.read() == {"lần": 19}
|
||||
assert list(tmp_path.glob("*.tmp")) == []
|
||||
|
||||
|
||||
def test_giu_nguyen_tieng_viet_khong_escape(tmp_path):
|
||||
"""config.py hiện dùng ensure_ascii=False — giữ nguyên để file đọc được
|
||||
bằng mắt và git diff không thành một đống \\uXXXX."""
|
||||
path = tmp_path / "vi.json"
|
||||
AtomicJsonFile(path).write({"tên": "Nguyễn Văn Đức"})
|
||||
raw = path.read_text(encoding="utf-8")
|
||||
assert "Nguyễn Văn Đức" in raw
|
||||
assert "\\u" not in raw
|
||||
|
||||
|
||||
def test_tao_thu_muc_cha_neu_chua_co(tmp_path):
|
||||
f = AtomicJsonFile(tmp_path / "sâu" / "hơn" / "nữa" / "c.json")
|
||||
f.write({"ok": True})
|
||||
assert f.read() == {"ok": True}
|
||||
|
||||
|
||||
def test_json_ghi_ra_doc_duoc_bang_thu_vien_chuan(tmp_path):
|
||||
path = tmp_path / "c.json"
|
||||
AtomicJsonFile(path).write({"n": [1, 2, {"m": None}]})
|
||||
assert json.loads(path.read_text(encoding="utf-8")) == {"n": [1, 2, {"m": None}]}
|
||||
@@ -0,0 +1,155 @@
|
||||
"""JsonConfigRepository — R02-T02.
|
||||
|
||||
Hai nhóm bài:
|
||||
* **round-trip** — ghi rồi nạp lại phải ra đúng thứ đã ghi (cột nghiệm thu
|
||||
của plan.md cho ngày 22-23/08)
|
||||
* **đường A** — ``provider_conf()`` vẫn trả ``api_key``, nhưng file JSON
|
||||
trên đĩa thì không có, để qua CASAN Check 1
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from cowork_local.infrastructure.config.config_repository import ConfigRepository
|
||||
from cowork_local.infrastructure.config.json_config_repository import (
|
||||
JsonConfigRepository,
|
||||
)
|
||||
from cowork_local.tests.fakes.fake_config import FakeSecretStore
|
||||
|
||||
DEFAULTS = {
|
||||
"active_provider": "ollama",
|
||||
"providers": {
|
||||
"ollama": {"base_url": "http://localhost:11434/v1", "model": "llama3",
|
||||
"api_key": "ollama"},
|
||||
"openai": {"base_url": "https://api.openai.com/v1", "model": "gpt-4o-mini",
|
||||
"api_key": ""},
|
||||
},
|
||||
"theme": "dark", "language": "vi", "shared_dir": "",
|
||||
"routing": {"mode": "off"}, "auth": {}, "agent_security": {},
|
||||
"tools_disabled": [], "history": {}, "cowork": {}, "ms365": {},
|
||||
}
|
||||
|
||||
|
||||
def _repo(tmp_path, secrets=None):
|
||||
return JsonConfigRepository(tmp_path / "config.json", secrets=secrets,
|
||||
defaults=DEFAULTS, env_overrides=lambda d: d)
|
||||
|
||||
|
||||
def test_khop_hop_dong(tmp_path):
|
||||
assert isinstance(_repo(tmp_path), ConfigRepository)
|
||||
|
||||
|
||||
def test_chua_co_file_thi_dung_mac_dinh(tmp_path):
|
||||
cfg = _repo(tmp_path)
|
||||
assert cfg.active_provider == "ollama"
|
||||
assert cfg.theme == "dark"
|
||||
|
||||
|
||||
def test_round_trip(tmp_path):
|
||||
cfg = _repo(tmp_path)
|
||||
cfg.set_theme("light")
|
||||
cfg.set_language("en")
|
||||
cfg.set_active_provider("openai")
|
||||
cfg.set_tool_enabled("run_command", False)
|
||||
cfg.save()
|
||||
|
||||
lai = _repo(tmp_path)
|
||||
assert lai.theme == "light"
|
||||
assert lai.language == "en"
|
||||
assert lai.active_provider == "openai"
|
||||
assert lai.tools_disabled == ["run_command"]
|
||||
|
||||
|
||||
def test_gia_tri_luu_trong_file_trum_len_mac_dinh_nhung_giu_phan_con_thieu(tmp_path):
|
||||
"""Trộn sâu: file cũ thiếu khoá mới thì lấy mặc định, không mất phần cũ."""
|
||||
(tmp_path / "config.json").write_text(
|
||||
json.dumps({"theme": "light", "providers": {"openai": {"model": "gpt-5"}}}),
|
||||
encoding="utf-8")
|
||||
cfg = _repo(tmp_path)
|
||||
assert cfg.theme == "light" # từ file
|
||||
assert cfg.language == "vi" # từ mặc định
|
||||
assert cfg.provider_conf("openai")["model"] == "gpt-5" # từ file
|
||||
assert "api.openai.com" in cfg.provider_conf("openai")["base_url"] # mặc định
|
||||
|
||||
|
||||
# ---- đường A: khoá vào kho bí mật, nhưng dict vẫn có ------------------------
|
||||
|
||||
def test_provider_conf_van_tra_api_key_sau_khi_chuyen_vao_kho(tmp_path):
|
||||
"""Điểm mấu chốt của quyết định A: 5 nơi đọc conf['api_key'] không đổi."""
|
||||
secrets = FakeSecretStore()
|
||||
cfg = _repo(tmp_path, secrets)
|
||||
cfg.set_api_key("openai", "sk-that-bi-mat")
|
||||
|
||||
assert cfg.provider_conf("openai")["api_key"] == "sk-that-bi-mat"
|
||||
|
||||
|
||||
def test_khoa_khong_bao_gio_nam_tren_dia(tmp_path):
|
||||
"""Điều kiện qua CASAN Check 1."""
|
||||
secrets = FakeSecretStore()
|
||||
cfg = _repo(tmp_path, secrets)
|
||||
cfg.set_api_key("openai", "sk-that-bi-mat")
|
||||
cfg.save()
|
||||
|
||||
raw = (tmp_path / "config.json").read_text(encoding="utf-8")
|
||||
assert "sk-that-bi-mat" not in raw
|
||||
assert secrets.get("provider:openai") == "sk-that-bi-mat"
|
||||
|
||||
|
||||
def test_sua_dict_tra_ve_khong_lam_ban_cau_hinh(tmp_path):
|
||||
"""provider_conf trả bản sao — nếu trả tham chiếu thì khoá vừa ghép vào sẽ
|
||||
lẫn ngược vào self.data rồi theo save() xuống đĩa."""
|
||||
secrets = FakeSecretStore()
|
||||
cfg = _repo(tmp_path, secrets)
|
||||
cfg.set_api_key("openai", "sk-bi-mat")
|
||||
|
||||
conf = cfg.provider_conf("openai")
|
||||
conf["model"] = "bị sửa bậy"
|
||||
cfg.save()
|
||||
|
||||
raw = (tmp_path / "config.json").read_text(encoding="utf-8")
|
||||
assert "bị sửa bậy" not in raw
|
||||
assert "sk-bi-mat" not in raw
|
||||
|
||||
|
||||
def test_khong_co_kho_bi_mat_thi_van_chay_nhu_cu(tmp_path):
|
||||
"""Máy không có keyring: hành vi lùi về đúng như config.py hôm nay."""
|
||||
cfg = _repo(tmp_path, secrets=None)
|
||||
cfg.set_api_key("openai", "sk-nam-trong-file")
|
||||
cfg.save()
|
||||
|
||||
assert cfg.provider_conf("openai")["api_key"] == "sk-nam-trong-file"
|
||||
raw = (tmp_path / "config.json").read_text(encoding="utf-8")
|
||||
assert "sk-nam-trong-file" in raw # đúng như cũ, có đánh đổi rõ ràng
|
||||
|
||||
|
||||
# ---- giữ nguyên hành vi cũ --------------------------------------------------
|
||||
|
||||
def test_ms365_unlocked_khong_bao_gio_xuong_dia(tmp_path):
|
||||
cfg = _repo(tmp_path)
|
||||
cfg.data["ms365"]["unlocked"] = True
|
||||
cfg.save()
|
||||
|
||||
raw = json.loads((tmp_path / "config.json").read_text(encoding="utf-8"))
|
||||
assert raw["ms365"]["unlocked"] is False
|
||||
assert cfg.data["ms365"]["unlocked"] is True # trong bộ nhớ vẫn giữ
|
||||
|
||||
assert _repo(tmp_path).data["ms365"]["unlocked"] is False
|
||||
|
||||
|
||||
def test_ghi_hong_giua_chung_khong_lam_mat_cau_hinh(tmp_path, monkeypatch):
|
||||
"""Thừa hưởng từ AtomicJsonFile — kiểm lại ở tầng này cho chắc."""
|
||||
import os
|
||||
|
||||
cfg = _repo(tmp_path)
|
||||
cfg.set_theme("light")
|
||||
cfg.save()
|
||||
|
||||
monkeypatch.setattr(os, "replace",
|
||||
lambda *a, **k: (_ for _ in ()).throw(OSError("mất điện")))
|
||||
cfg.set_theme("hỏng")
|
||||
with pytest.raises(OSError):
|
||||
cfg.save()
|
||||
|
||||
assert _repo(tmp_path).theme == "light"
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user