Team Hoa, EPIC R08 (UI/Application Separation) - Team Hoa scope only
(R08-T11 -> T14; R08-T01->T10 belong to Team Duy/Team Nam).
- R08-T11: ui/schedule_task_tab.py (795 lines) -> presentation/scheduling/
{kanban_board_widget,calendar_view_widget,ai_task_creator_dialog,
ai_task_import_dialog,run_history_dialog}.py + schedule_task_tab.py
shell. Kanban CRUD/drag-drop now goes through
application/scheduling/task_application_service.py (R07-T04) instead of
~30 lines of inline if/elif per drag target.
- R08-T12: ui/folder_tab.py (1587 lines, the largest of the four) ->
presentation/folder/{workspace_file_tree,document_preview_manager,
code_editor,office_document_renderer,ai_file_editor_dialog,
ai_edit_model_resolver,ai_edit_pipeline}.py + folder_tab.py shell.
Closes the R06-T05 loop: FileWorkspaceService existed since R06 with
zero production call sites (confirmed by grep); every plain-text write
(save/create/write_content) now goes through it, gaining path
containment and a Python-syntax warning the original code never had.
Pure helpers (_read_text, _is_probably_text, _pptx_available,
_split_code_block, _parse_ai_output) moved to
application/workspaces/{file_preview_helpers,ai_edit_output}.py.
- R08-T13: ui/dashboard_tab.py (437 lines) -> presentation/dashboard/
{token_usage_card_widget,usage_chart_widget,habits_widget}.py +
dashboard_tab.py shell, backed by a new
application/monitoring/dashboard_query_service.py (pricing/period/
summary queries the three widgets used to each recompute separately).
Directory-ownership note left in the checklist for Team Nam.
- R08-T14: ui/structure_graph_view.py (1035 lines) ->
presentation/graph/{graph_scene_items,graph_renderer,
graph_messages_view,graph_qa_widget}.py + structure_graph_view.py
shell. Extraction helpers (_pdf_to_markdown, _extract_file_contents)
moved to application/workspaces/graph_index_service.py (pure Python).
Renderer and Q&A panel talk only through signals
(node_selected/graph_rendered/raw_json_ready/project_changed) - neither
imports the other.
- presentation/shared/web_engine_support.py: HAS_WEB_ENGINE, previously
duplicated (folder_tab imported it FROM structure_graph_view.py) - now
one shared flag instead of one screen importing another screen's module.
All four old ui/*.py files deleted; app.py and ui/workspace_tab.py updated
to the new import paths (each god-file only had 1-2 real construction
sites, so import sites were updated directly rather than kept as a
strangler-fig shim - unlike core/tools.py at R05, which had dozens).
pytest: 377 pass (+94 vs the R07 baseline of 328; same 4 pre-existing
failures as the R05/R06 baseline, unrelated to this work).
scripts/check_imports.py: PASS. python -c "import cowork_local.app": OK.
Every new file < 400 lines (largest: graph_renderer.py, 391).
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
92 lines
3.6 KiB
Python
92 lines
3.6 KiB
Python
"""Temporary file-content extraction for Graph-RAG Q&A (R08-T14, moved out
|
|
of ``ui/structure_graph_view.py`` — that file's module-level
|
|
``_pdf_to_markdown``/``_extract_file_contents``, lines 964-1034 of the
|
|
original 1035-line file). Runs inside the ask worker's job function so the
|
|
answer is synthesized from real file content, not just the graph structure.
|
|
|
|
Pure Python: no Qt. Best-effort throughout (never raises) — a failed
|
|
extraction degrades to "no content for this file", not a broken Q&A turn.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
from typing import Dict, List, Optional, Tuple
|
|
|
|
|
|
def pdf_to_markdown(pdf_path: str, out_dir: str) -> Optional[str]:
|
|
"""Convert a PDF to Markdown with opendataloader-pdf when available
|
|
(richer structure than a plain text dump). Best-effort — returns None
|
|
if the package isn't installed or the call fails, so the caller falls
|
|
back to ``core/doc_extract.py``."""
|
|
try:
|
|
import opendataloader_pdf # optional; auto-installed elsewhere if present
|
|
except Exception: # noqa: BLE001
|
|
try:
|
|
from cowork_local.core.deps import ensure_module
|
|
if ensure_module("opendataloader_pdf", "opendataloader-pdf") is None:
|
|
return None
|
|
import opendataloader_pdf # noqa: F811
|
|
except Exception: # noqa: BLE001
|
|
return None
|
|
out = Path(out_dir)
|
|
out.mkdir(parents=True, exist_ok=True)
|
|
for call in (
|
|
lambda: opendataloader_pdf.convert(input_path=[str(pdf_path)], output_dir=str(out),
|
|
generate_markdown=True),
|
|
lambda: opendataloader_pdf.convert(input_path=str(pdf_path), output_dir=str(out)),
|
|
lambda: opendataloader_pdf.convert(str(pdf_path), str(out)),
|
|
):
|
|
try:
|
|
call()
|
|
break
|
|
except TypeError:
|
|
continue
|
|
except Exception: # noqa: BLE001
|
|
return None
|
|
mds = list(out.rglob(Path(pdf_path).stem + "*.md")) or list(out.rglob("*.md"))
|
|
for md in mds:
|
|
try:
|
|
return md.read_text(encoding="utf-8", errors="replace")
|
|
except OSError:
|
|
continue
|
|
return None
|
|
|
|
|
|
def extract_file_contents(paths: List[str], cache: Dict[str, str], tmp_dir: str,
|
|
max_files: int = 15, max_total: int = 120_000
|
|
) -> Tuple[str, Dict[str, str]]:
|
|
"""Read the ACTUAL content of ``paths`` (PDF -> markdown via
|
|
opendataloader when available, else ``doc_extract`` for office/pdf/
|
|
text). Returns ``(block, cache)`` — ``block`` is the concatenated
|
|
content for the prompt (bounded), ``cache`` maps path -> text for
|
|
reuse. Never raises."""
|
|
from cowork_local.core import doc_extract
|
|
|
|
cache = dict(cache or {})
|
|
parts, total = [], 0
|
|
for p in paths[:max_files]:
|
|
if total >= max_total:
|
|
break
|
|
text = cache.get(p)
|
|
if text is None:
|
|
try:
|
|
if Path(p).suffix.lower() == ".pdf":
|
|
text = pdf_to_markdown(p, tmp_dir)
|
|
if not text:
|
|
text, _n = doc_extract.extract_text(p)
|
|
else:
|
|
text, _n = doc_extract.extract_text(p)
|
|
except Exception: # noqa: BLE001
|
|
text = ""
|
|
cache[p] = text or ""
|
|
text = cache.get(p) or ""
|
|
if not text:
|
|
continue
|
|
chunk = text[: max(0, max_total - total)]
|
|
total += len(chunk)
|
|
parts.append(f'--- {Path(p).name} ({p}) ---\n{chunk}')
|
|
return ("\n\n".join(parts), cache)
|
|
|
|
|
|
__all__ = ["pdf_to_markdown", "extract_file_contents"]
|