From f23fb98953da8fe823f009db15097a680d8641fc Mon Sep 17 00:00:00 2001 From: thanhnv Date: Sat, 18 Jul 2026 11:29:37 +0700 Subject: [PATCH] fix: validate and size governed patch outputs --- .../scripts/bash/goal-orchestrator.py | 51 +++++++++++++++---- .../casan-harness/scripts/bash/model-call.py | 17 +++++-- .../tests/goal-patch-workflow-tests.py | 6 +++ 3 files changed, 61 insertions(+), 13 deletions(-) diff --git a/packages/casan-harness/scripts/bash/goal-orchestrator.py b/packages/casan-harness/scripts/bash/goal-orchestrator.py index 89937df..72bdaba 100644 --- a/packages/casan-harness/scripts/bash/goal-orchestrator.py +++ b/packages/casan-harness/scripts/bash/goal-orchestrator.py @@ -267,6 +267,23 @@ def extract_patch(text: str) -> str: return patch +def validate_patch_check(patch: str) -> None: + """Require a candidate diff to apply before it can pass H2 or H3.""" + with tempfile.NamedTemporaryFile("w", encoding="utf-8", suffix=".patch") as handle: + handle.write(patch) + handle.flush() + try: + check = subprocess.run( + ["git", "apply", "--check", "--whitespace=error", handle.name], + cwd=ROOT, capture_output=True, text=True, timeout=30, + ) + except FileNotFoundError as error: + raise ValueError("goal_patch_validator_unavailable") from error + if check.returncode != 0: + detail = (check.stderr or check.stdout or "git_apply_check_failed").strip().replace("\n", " ")[:160] + raise ValueError(f"goal_patch_check_failed:{detail}") + + def validate_write_output(text: str) -> str: """Fail at the producing harness when a write-intent reply is not a diff. @@ -274,7 +291,9 @@ def validate_write_output(text: str) -> str: reviewer has completed, while ensuring H2/H3 accurately identify a model that violated the required output contract. """ - return extract_patch(text) + patch = extract_patch(text) + validate_patch_check(patch) + return patch def patch_repair_attempts() -> int: @@ -286,6 +305,15 @@ def patch_repair_attempts() -> int: return min(max(configured, 0), 1) +def patch_output_tokens() -> int: + """Keep a multi-hunk patch from being cut at the generic chat limit.""" + try: + configured = int(os.environ.get("CASAN_GOAL_PATCH_MAX_OUTPUT_TOKENS", "6000")) + except ValueError: + configured = 6000 + return min(max(configured, 512), 8192) + + def patch_repair_models(primary_model: str): """Use direct cloud -> gateway -> local order for one bounded H2 recovery.""" configured = [item.strip() for item in os.environ.get("CASAN_GOAL_PATCH_REPAIR_MODELS", "").split(",") if item.strip()] @@ -309,7 +337,7 @@ def provider_for_model(model: str) -> str: return "model" -def repair_write_output(model: str, original_prompt: str, invalid_output: str): +def repair_write_output(model: str, original_prompt: str, invalid_output: str, contract_error="goal_patch_missing"): """Ask the producing model to repair format only, without relaxing H2. A write-intent goal may never advance to approval without a checked unified @@ -321,14 +349,14 @@ def repair_write_output(model: str, original_prompt: str, invalid_output: str): if patch_repair_attempts() == 0: return False, "", {}, "goal_patch_missing" repair_prompt = ( - "Your previous response violated the required write-output contract because it did not contain a complete unified git diff. " - "Return ONLY one complete patch inside a ```diff fence. The first non-empty line inside the fence MUST be `diff --git `. " - "Do not explain, plan, summarize, use placeholders, or perform side effects. Preserve the original objective and workspace restrictions.\n\n" + f"Your previous response violated the required write-output contract: {contract_error}. " + "Return ONLY one complete, applicable unified git diff inside a ```diff fence. The first non-empty line inside the fence MUST be `diff --git `. " + "Every hunk must be complete and the patch must pass `git apply --check`. Do not emit `index` lines, placeholder hashes, commentary, plans, summaries, or side effects. Preserve the original objective and workspace restrictions.\n\n" f"ORIGINAL CONTRACT:\n{original_prompt}\n\nPREVIOUS INVALID RESPONSE:\n{invalid_output[:8000]}" ) candidates = patch_repair_models(model) candidate = candidates[0] if candidates else model - ok, output, metadata, reason = call_model(candidate, repair_prompt, candidate.startswith(("openai:", "anthropic:", "openai-compatible:"))) + ok, output, metadata, reason = call_model(candidate, repair_prompt, candidate.startswith(("openai:", "anthropic:", "openai-compatible:")), max_output_tokens=patch_output_tokens()) metadata = dict(metadata) metadata["repair_model"] = candidate if not ok: @@ -438,7 +466,7 @@ def scan(text: str, mode: str): return result.returncode == 0, safe -def call_model(model: str, prompt: str, cloud: bool, timeout_seconds=None): +def call_model(model: str, prompt: str, cloud: bool, timeout_seconds=None, max_output_tokens=None): if not model: return False, "", {}, "model_unconfigured" with tempfile.TemporaryDirectory() as directory: @@ -452,7 +480,10 @@ def call_model(model: str, prompt: str, cloud: bool, timeout_seconds=None): # ingest the brief and produce a bounded plan. The outer goal timeout # remains the hard ceiling; this only raises the router's 60s default. environment.setdefault("CASAN_MODEL_TIMEOUT_SEC", os.environ.get("CASAN_GOAL_LOCAL_TIMEOUT_SEC", "240") if not cloud else "120") - environment.setdefault("CASAN_MODEL_GENERATE_MAX_TOKENS", os.environ.get("CASAN_GOAL_MAX_OUTPUT_TOKENS", "1400")) + configured_output = os.environ.get("CASAN_GOAL_MAX_OUTPUT_TOKENS", "1400") + if max_output_tokens is not None: + configured_output = str(max_output_tokens) + environment["CASAN_MODEL_GENERATE_MAX_TOKENS"] = configured_output timeout = max(1, int(timeout_seconds or os.environ.get("CASAN_GOAL_MODEL_TIMEOUT", "300"))) environment["CASAN_MODEL_TIMEOUT_SEC"] = str(min( int(environment.get("CASAN_MODEL_TIMEOUT_SEC", timeout)), timeout @@ -728,7 +759,7 @@ def run(job_path: str) -> int: repair_provider = provider_for_model(repair_model) stage(job_path, "local-worker", "running", "Repairing invalid patch output contract", repair_provider, repair_model) emit(goal_id, "H2-tool", "running", "Worker is repairing patch output contract", {"reason": str(first_error), "max_attempts": patch_repair_attempts(), "repair_provider": repair_provider, "repair_model": repair_model}) - repaired, repaired_output, repaired_meta, repair_reason = repair_write_output(local_model, local_prompt, safe_local) + repaired, repaired_output, repaired_meta, repair_reason = repair_write_output(local_model, local_prompt, safe_local, str(first_error)) local_meta = merge_usage(local_meta, repaired_meta) if not repaired: reason = repair_reason @@ -754,7 +785,7 @@ def run(job_path: str) -> int: stage(job_path, "cloud-reviewer", "running", "Independent reviewer is challenging and improving the local solution", job.get("cloud_provider", ""), cloud_model) emit(goal_id, "H3-eval", "running", "Cloud reviewer is evaluating the local solution", {"provider": job.get("cloud_provider", ""), "model": cloud_model}) reviewer_contract = ( - "Return ONLY the corrected, complete unified git diff inside a ```diff fence. Preserve valid implementation details, fix defects, and include no prose outside the diff. " + "Return ONLY the corrected, complete unified git diff inside a ```diff fence. Preserve valid implementation details, fix defects, and include no prose or `index` lines outside the diff. Every hunk must be complete and pass `git apply --check`. " if write_intent else "Return one final actionable solution with ordered steps and acceptance checks. Respond in the same language as the objective. " ) diff --git a/packages/casan-harness/scripts/bash/model-call.py b/packages/casan-harness/scripts/bash/model-call.py index 80a28c3..9212f8f 100755 --- a/packages/casan-harness/scripts/bash/model-call.py +++ b/packages/casan-harness/scripts/bash/model-call.py @@ -56,6 +56,17 @@ PROVIDER_LOG = os.path.join(REPO_ROOT, ".specify/logs/level5/provider-usage.json REQUEST_TIMEOUT = max(1, int(os.environ.get("CASAN_MODEL_TIMEOUT_SEC", "60"))) +def generation_max_tokens(role): + """Return a bounded generation budget shared by every model provider.""" + if role in ("classify", "judge"): + return 16 + try: + configured = int(os.environ.get("CASAN_MODEL_GENERATE_MAX_TOKENS", "1400")) + except ValueError: + configured = 1400 + return min(max(configured, 64), 8192) + + def enforce_call_budget(): """Fail closed when a run exceeds CASAN_MODEL_MAX_CALLS. The count is tracked in CASAN_MODEL_CALL_COUNTER_FILE so it spans the many per-step model-call processes @@ -205,7 +216,7 @@ def call_openai(model_name, prompt, role): "model": model_name, "messages": [{"role": "user", "content": prompt}], "temperature": 0 if role in ("classify", "judge") else 0.2, - "max_tokens": 16 if role in ("classify", "judge") else 512, + "max_tokens": generation_max_tokens(role), } data = json.dumps(body).encode() req = urllib.request.Request( @@ -272,7 +283,7 @@ def call_openai_compatible(model_name, prompt, role): "model": model_name, "messages": [{"role": "user", "content": prompt}], "temperature": 0 if role in ("classify", "judge") else 0.2, - "max_tokens": 16 if role in ("classify", "judge") else 512, + "max_tokens": generation_max_tokens(role), } data = json.dumps(body).encode() req = urllib.request.Request( @@ -335,7 +346,7 @@ def call_anthropic(model_name, prompt, role): url = f"https://{host}/v1/messages" body = { "model": model_name, - "max_tokens": 16 if role in ("classify", "judge") else 512, + "max_tokens": generation_max_tokens(role), "messages": [{"role": "user", "content": prompt}], } data = json.dumps(body).encode() diff --git a/packages/casan-harness/tests/goal-patch-workflow-tests.py b/packages/casan-harness/tests/goal-patch-workflow-tests.py index 91ca9c8..8b52b2e 100644 --- a/packages/casan-harness/tests/goal-patch-workflow-tests.py +++ b/packages/casan-harness/tests/goal-patch-workflow-tests.py @@ -38,6 +38,11 @@ class GoalPatchWorkflowTests(unittest.TestCase): value = ORCHESTRATOR.extract_patch("```diff\ndiff --git a/apps/okr/frontend/a.ts b/apps/okr/frontend/a.ts\n--- a/apps/okr/frontend/a.ts\n+++ b/apps/okr/frontend/a.ts\n@@ -1 +1 @@\n-a\n+b\n```") self.assertTrue(value.startswith("diff --git")) + def test_write_contract_rejects_a_truncated_diff_before_h3(self): + truncated = "diff --git a/apps/okr/frontend/a.ts b/apps/okr/frontend/a.ts\n--- a/apps/okr/frontend/a.ts\n+++ b/apps/okr/frontend/a.ts\n@@ -1 +1 @@\n-a\n" + with self.assertRaisesRegex(ValueError, "goal_patch_check_failed"): + ORCHESTRATOR.validate_write_output(truncated) + def test_invalid_write_output_gets_one_bounded_repair_attempt(self): repaired = "```diff\ndiff --git a/apps/okr/frontend/a.ts b/apps/okr/frontend/a.ts\n--- a/apps/okr/frontend/a.ts\n+++ b/apps/okr/frontend/a.ts\n@@ -1 +1 @@\n-a\n+b\n```" with patch.dict(os.environ, {"CASAN_GOAL_PATCH_REPAIR_ATTEMPTS": "1"}, clear=False), \ @@ -68,6 +73,7 @@ class GoalPatchWorkflowTests(unittest.TestCase): self.assertEqual(reason, "ok") self.assertEqual(usage["repair_model"], "openai:gpt-4o-mini") self.assertEqual(call.call_args.args[0], "openai:gpt-4o-mini") + self.assertEqual(call.call_args.kwargs["max_output_tokens"], ORCHESTRATOR.patch_output_tokens()) def test_patch_outside_workspace_is_denied(self): job = {"workspace": {"context_roots": ["apps/okr/frontend"]}}