Người dùng phản ánh dán nội dung vào chat cho câu trả lời tốt hơn đính kèm cùng nội dung đó. Nội dung KHÔNG bị cắt (giới hạn 2 triệu ký tự) — nguyên nhân là loãng và trùng: _augment luôn quét thêm [Workspace files] và [Project files], và tệp đính kèm nếu nằm trong thư mục workspace sẽ đi vào prompt HAI lần. Với tài liệu dài, bản thứ hai vừa nhân đôi ngữ cảnh vừa khiến model không biết bản nào là bản được hỏi. Lọc trùng theo đường dẫn đã giải quyết, và đổi nhãn để nói rõ tệp đính kèm là CHỦ THỂ CHÍNH còn tệp thư mục chỉ là ngữ cảnh phụ. Dòng cảnh báo "N tệp không nạp được" đếm TRƯỚC khi lọc trùng, nếu không nó báo sai. Cùng lượt, hai chỗ bỏ thông tin không cần thiết: - gỡ dòng "provider · model" ngay sau chữ "Cowork" — nó lặp lại thứ bộ chọn provider ở thanh trên đang hiển thị, mà chiếm chỗ đắt nhất trên thanh công cụ; - không báo "đang dùng <provider>" ở thanh trạng thái khi đổi provider — bộ chọn nằm ngay trên màn hình và đã hiện thứ người dùng vừa tự chọn. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
244 lines
12 KiB
Python
244 lines
12 KiB
Python
"""Tệp đính kèm của một lượt chat — R08-T03.
|
|
|
|
Đọc nội dung tệp người dùng kèm vào rồi ghép vào câu hỏi. Ba thứ đáng
|
|
chú ý:
|
|
|
|
* ``_enforce_attachment_security`` chạy TRƯỚC khi nội dung vào ngữ cảnh
|
|
model — đây là một trong ba tầng kiểm của R09.
|
|
* ``_attach_char_limit`` cắt bớt tệp quá dài; không cắt thì một tệp log
|
|
vài chục MB đủ làm hỏng cả lượt.
|
|
* Kèm cả thư mục thì chỉ lấy DANH SÁCH tệp, không đọc nội dung từng cái.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
from typing import List
|
|
from PySide6.QtCore import Qt
|
|
from ...core.worker import AgentWorker
|
|
from ...i18n import tr
|
|
from ...ui.osutil import is_image
|
|
|
|
|
|
class AttachmentMixin:
|
|
"""Trộn vào ChatPanel."""
|
|
|
|
def _on_attachments_added(self, paths: List[str]) -> None:
|
|
# Push attachments into the Input box as soon as they're attached.
|
|
"""Tệp vừa được đính kèm: đưa luôn vào mục "Tệp đầu vào" để người dùng thấy ngay."""
|
|
for p in paths:
|
|
self.input_section.add(p)
|
|
|
|
def _on_attachment_removed(self, path: str) -> None:
|
|
# A file added by mistake was removed in the composer — drop it from the
|
|
# Input panel too (only matters before the message is sent).
|
|
"""Gỡ tệp đính kèm nhầm khỏi cả mục "Tệp đầu vào" (chỉ có ý nghĩa trước khi gửi)."""
|
|
self.input_section.remove(path)
|
|
|
|
def _attach_char_limit(self) -> int:
|
|
"""Per-file content cap (characters) from the Settings token limit
|
|
(~4 chars/token)."""
|
|
try:
|
|
tokens = int(self.ctx.config.data.get("attachments", {}).get("max_tokens", 500000))
|
|
except (TypeError, ValueError):
|
|
tokens = 500000
|
|
return max(1000, tokens) * 4
|
|
|
|
def _augment(self, text: str, attachments: List[str], notify=None) -> str:
|
|
"""Embed attachment paths AND their extracted contents into the prompt so
|
|
the agent actually reads and analyses each attached file.
|
|
|
|
Additionally, scans the workspace/output folder for existing files and
|
|
loads them as input data so the agent can read/process them automatically.
|
|
|
|
``notify``, if given, is called with UI-visible events (a live "reading
|
|
page X/Y" progress notice, and a warning when a file's content could not
|
|
be read) instead of failures being silently handed to the model as an
|
|
opaque inline note."""
|
|
has_attachments = bool(attachments)
|
|
limit = self._attach_char_limit()
|
|
lines = [text] if text else []
|
|
|
|
# --- User-attached files ---
|
|
# Đường dẫn đã giải quyết của các tệp đính kèm, để vòng quét thư mục
|
|
# phía sau không gửi lại chính chúng một lần nữa.
|
|
da_dinh_kem = set()
|
|
if has_attachments:
|
|
lines.append(
|
|
"\n[Attachments] — the user attached these files for THIS request. "
|
|
"They are the PRIMARY subject: read them in full and base the answer "
|
|
"on them. Anything listed further below is background context only.")
|
|
for p in attachments:
|
|
try:
|
|
da_dinh_kem.add(str(Path(p).resolve()))
|
|
except OSError:
|
|
pass
|
|
lines.extend(self._read_one_attachment(p, limit, notify))
|
|
|
|
# --- Auto-load existing workspace/output folder files as input data ---
|
|
# This is what makes "📁 Chọn thư mục khác" useful as an INPUT folder
|
|
# too: every file already in the chosen folder is read and embedded so
|
|
# the agent can act on their contents without manual attaching.
|
|
workspace = self.workspace_dir()
|
|
max_files = int(self.ctx.config.data.get("attachments", {})
|
|
.get("max_files", 10) or 0)
|
|
if workspace is not None:
|
|
lines.extend(self._folder_input_lines(
|
|
workspace,
|
|
"[Workspace files] — other files that happen to sit in the output "
|
|
"folder. Background context; do NOT let them displace the "
|
|
"attached files or the user's own question:",
|
|
limit, max_files, notify, da_dinh_kem))
|
|
|
|
# --- Project knowledge (Claude-Projects style) ---
|
|
# Only scanned separately when it's a DIFFERENT folder from the
|
|
# session's own workspace — for Cowork the two are now the same
|
|
# folder (a project has one shared workspace, no per-thread
|
|
# sub-folder), so this never double-scans the same directory.
|
|
knowledge = self.project_knowledge_dir()
|
|
if knowledge is not None and knowledge != workspace:
|
|
lines.extend(self._folder_input_lines(
|
|
knowledge,
|
|
"[Project files] — shared knowledge of this project. Background "
|
|
"context; do NOT let them displace the attached files or "
|
|
"the user's own question:",
|
|
limit, max_files, notify, da_dinh_kem))
|
|
|
|
return "\n".join(lines)
|
|
|
|
def _folder_input_lines(self, folder: Path, header: str, limit: int,
|
|
max_files: int, notify=None, skip=frozenset()) -> list:
|
|
"""Embed a folder's readable files into the prompt — recursing into
|
|
every sub-folder, any depth, not just the top level, so files placed
|
|
in nested folders are read and processed too (same per-message file
|
|
cap as manual attachments — Settings → Attachments → max files;
|
|
0 = unlimited — so a folder with dozens of files can't blow the
|
|
context window)."""
|
|
from pathlib import Path as _P
|
|
|
|
from ...core.doc_extract import find_input_files
|
|
|
|
out: list = []
|
|
shown, total = find_input_files(folder, self._INPUT_EXTS, max_files)
|
|
# Bo qua tep nguoi dung DA dinh kem tuong minh. Tep dinh kem thuong nam
|
|
# ngay trong thu muc workspace, nen khong loc thi cung mot tai lieu di vao
|
|
# prompt HAI lan: mot lan duoi [Attachments], mot lan duoi [Workspace
|
|
# files]. Voi tai lieu dai, ban thu hai vua nhan doi ngu canh vua khien
|
|
# model khong biet ban nao la ban duoc hoi.
|
|
# Số tệp thư mục này thực sự trả về, ĐO TRƯỚC khi lọc trùng: dòng cảnh
|
|
# báo bên dưới nói về giới hạn mỗi lượt, nên đếm cả tệp bị lọc vì đã
|
|
# đính kèm sẽ báo sai là "không nạp được".
|
|
so_lay_duoc = len(shown)
|
|
if skip:
|
|
shown = [f for f in shown if str(_P(f).resolve()) not in skip]
|
|
if shown:
|
|
out.append("\n" + header)
|
|
for f in shown:
|
|
out.extend(self._read_one_attachment(str(f), limit, notify))
|
|
if total > so_lay_duoc:
|
|
skipped = total - so_lay_duoc
|
|
out.append(f"…({skipped} more files in the folder were not "
|
|
"loaded — per-message attachment limit; mention a "
|
|
"file by name if the user asks about it)")
|
|
if notify is not None:
|
|
notify({"type": "notice", "level": "warning",
|
|
"text": tr("chat.workspace_files_capped",
|
|
shown=len(shown), total=total)})
|
|
return out
|
|
|
|
def _read_one_attachment(self, path: str, limit: int, notify=None) -> list:
|
|
"""Read and format one attachment/workspace file. Returns list of lines.
|
|
|
|
Handles every file type: images (noted with path), MS Office / PDF /
|
|
OpenDocument / text (extracted), and ZIP archives — which are auto-
|
|
extracted into the workspace and their contents read + processed."""
|
|
name = Path(path).name
|
|
result = []
|
|
if is_image(path):
|
|
result.append(f"- {name} (image at {path})")
|
|
return result
|
|
from ...core.doc_extract import is_zip
|
|
if is_zip(path):
|
|
result.extend(self._read_zip_attachment(path, name, limit, notify))
|
|
return result
|
|
|
|
def progress(page: int, total: int, _name=name) -> None:
|
|
"""Báo tiến độ trích nội dung, chỉ với tệp nhiều trang — tệp một trang thì dòng
|
|
tiến độ chỉ làm nhiễu.
|
|
"""
|
|
if notify is not None and total > 1:
|
|
notify({"type": "notice", "level": "progress",
|
|
"text": tr("chat.reading_progress", name=_name, page=page, total=total)})
|
|
|
|
content, note = self._read_attachment_text(path, progress=progress)
|
|
if content is None:
|
|
result.append(f"- {name} ({note}; located at {path})")
|
|
if notify is not None:
|
|
notify({"type": "notice", "level": "warning",
|
|
"text": tr("chat.attachment_failed", name=name, note=note)})
|
|
return result
|
|
self._enforce_attachment_security(name, content) # raises SecurityBlocked on a violation
|
|
extra = ""
|
|
if len(content) > limit:
|
|
content = content[:limit]
|
|
extra = f"\n…(truncated to ~{limit // 4} tokens)…"
|
|
result.append(f"- {name} ({path})")
|
|
result.append(f"\n--- Content of {name} ---\n{content}{extra}\n--- end of {name} ---")
|
|
return result
|
|
|
|
def _read_zip_attachment(self, path: str, name: str, limit: int, notify=None) -> list:
|
|
"""Auto-extract a .zip into the workspace and read+process its files, so
|
|
an attached archive is unpacked and its contents used automatically."""
|
|
from ...core.doc_extract import extract_archive
|
|
ws = self.workspace_dir()
|
|
dest = (Path(ws) if ws is not None else Path(path).parent) / Path(name).stem
|
|
files = extract_archive(path, dest)
|
|
result = [f"- {name} (archive) — extracted {len(files)} file(s) into the workspace at "
|
|
f"{dest}. Read/edit them there as needed."]
|
|
if self.workspace_dir() is not None:
|
|
self.output_changed.emit(str(self.workspace_dir())) # let the graph/folder refresh
|
|
max_files = int(self.ctx.config.data.get("attachments", {}).get("max_files", 10) or 0)
|
|
shown = files[:max_files] if max_files else files
|
|
for f in shown:
|
|
result.extend(self._read_one_attachment(str(f), limit, notify))
|
|
if max_files and len(files) > max_files:
|
|
result.append(f"- …and {len(files) - max_files} more file(s) in {dest} "
|
|
"(not inlined; open/read them from the workspace as needed).")
|
|
return result
|
|
|
|
def _enforce_attachment_security(self, filename: str, content: str) -> None:
|
|
"""Agent Security's attachment layer (Settings → 🛡 Agent Security) —
|
|
scans extracted file content for malicious payloads BEFORE it enters
|
|
the model's context. No-op when disabled. Raises SecurityBlocked
|
|
(propagates out of _augment → the worker job → AgentWorker.failed,
|
|
which the panel shows as a chat error) on a violation."""
|
|
sec = self.ctx.config.data.get("agent_security", {})
|
|
if not sec.get("enabled") or not sec.get("validate_attachments", True):
|
|
return
|
|
from ...core.agent_security import SecurityBlocked, combined_rules_text, validate_attachment
|
|
from ...core.agent_security_alert import notify_admin
|
|
|
|
rules_text = combined_rules_text(self.ctx.config)
|
|
verdict = validate_attachment(self.build_provider(), filename, content, rules_text)
|
|
if verdict.allowed:
|
|
return
|
|
notify_admin(self.ctx.config, verdict, detail=f"file: {filename}")
|
|
raise SecurityBlocked(verdict)
|
|
|
|
@staticmethod
|
|
def _read_attachment_text(path: str, progress=None):
|
|
"""Best-effort text extraction so the agent can read the attachment.
|
|
Returns (text, note); text is None when nothing readable was found.
|
|
|
|
Delegates to core.doc_extract, which parses docx/xlsx/pptx/odf directly
|
|
(stdlib, no extra packages), uses pypdf for PDFs (reporting per-page
|
|
``progress`` for multi-page files), and falls back to a headless
|
|
LibreOffice conversion for anything else."""
|
|
from ...core.doc_extract import extract_text
|
|
|
|
return extract_text(path, progress=progress)
|
|
|
|
def project_knowledge_dir(self):
|
|
"""Folder of project-level shared knowledge files (None = no project
|
|
knowledge). Overridden by the Cowork tab for non-default projects."""
|
|
return None
|