Files
CASAN/packages/casan-harness/scripts/bash/loop-convergence.py
T
thanhnvandClaude Opus 4.8 36a4812ef3 refactor(structure): promote app to repo root + remove redundant workspace cruft
Standard production layout: the OKR app (was nested under AINative_OKR_CASAN5/) is now
the repository root. No more wrapper directory.

- Promote AINative_OKR_CASAN5/* -> repo root (backend/ frontend/ packages/ apps/
  .specify/ docs/ infra/ nginx/ scripts/ + configs). Merge tool dirs: .gitea (kept the
  active deploy ci.yml, added harness-ci.yml + runbooks), .claude (agents/commands +
  launch.json), .github moved up.
- Remove redundant: 00_SUBMISSION_PACKAGE, scattered root notes (FPT_CASAN_Full.md,
  tu-tuong-casan.md, casan-tu-sinh..., casan_harness_assessment.md, source-review...,
  README_CASAN5_REFINED.md), casan-next-plans/ and optimize-docs/ (competition/planning
  artifacts — roadmap + design history preserved in git log / commit messages).
- Update all references to the old layout:
  - .gitea/workflows/{ci,harness-ci}.yml, .github/workflows/{ci,deploy}.yml:
    working-directory .; drop AINative_OKR_CASAN5/ prefix; .specify/{tests,scripts}
    -> packages/casan-harness/... (.specify/logs state kept)
  - .claude/launch.json, .gitea/*-runbook.md: path prefixes
  - CLAUDE.md, README.md: docs/input -> apps/okr/domain/input
  - policy-bundle.yaml: 8 policy paths -> packages/casan-harness/...; manifest re-signed
- secrets-scan.sh: fixture excludes -> new package/domain paths.

Full gate from the new root: PASS=64 FAIL=0 SKIP=3.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-08 13:26:36 +09:00

207 lines
7.0 KiB
Python

#!/usr/bin/env python3
"""CASAN No-progress / Oscillation Detector (Plan-17 Track 2, harness-owned).
Watches a run's per-step observations and decides whether the loop is CONVERGING,
STALLED (no forward progress), or OSCILLATING (repeating/thrashing actions). A
stalled or oscillating loop is halted or escalated (on_stall policy) instead of
being allowed to burn budget going nowhere.
Detection rules (thresholds from loop-policy.yaml, else strictest built-in):
* oscillation = the last N observations share one action_hash (stuck repeating),
* thrash = the last `thrash_window` steps alternate between exactly 2
action_hashes (A,B,A,B...),
* no-progress = the last W observations made zero forward progress
(progress never exceeded the running best).
Deny-by-default & fail-closed: absent/unknown policy => strictest thresholds
(detect sooner); a corrupt policy => ESCALATE/HALT, never silent CONVERGING.
Insufficient data (< window) => CONVERGING (the Budget Governor is the hard stop).
Usage:
loop-convergence.py observe --run-id R --step N --action-hash H --progress P
loop-convergence.py verdict --run-id R [--profile prod|dev]
Exit codes:
0 CONVERGING (or observe recorded ok)
3 HALT (STALLED/OSCILLATING with on_stall=halt, OR fail-closed error)
4 ESCALATE (STALLED/OSCILLATING with on_stall=escalate)
2 usage error
"""
import argparse
import json
import os
import sys
import loop_common as lc
def _emit(payload, stream=sys.stdout):
json.dump(payload, stream, ensure_ascii=False, indent=2, sort_keys=True)
stream.write("\n")
def _obs_path(run_id):
return os.path.join(lc.run_dir(run_id), "convergence.jsonl")
def _load_obs(run_id):
path = _obs_path(run_id)
rows = []
if os.path.isfile(path):
with open(path, encoding="utf-8") as fh:
for line in fh:
line = line.strip()
if not line:
continue
try:
rows.append(json.loads(line))
except ValueError:
# A corrupt observation stream cannot be trusted.
raise lc.PolicyError("observation_stream_corrupt")
rows.sort(key=lambda r: r.get("step", 0))
return rows
def cmd_observe(args):
path = _obs_path(args.run_id)
os.makedirs(os.path.dirname(path), exist_ok=True)
rec = {
"step": args.step,
"action_hash": args.action_hash,
"progress": args.progress,
"ts": lc.now_iso(),
}
with open(path, "a", encoding="utf-8") as fh:
fh.write(json.dumps(rec, ensure_ascii=False, sort_keys=True) + "\n")
_emit({"decision": "OBSERVED", "run_id": args.run_id, "step": args.step})
return 0
def _detect_oscillation(hashes, cfg):
n = cfg["oscillation_repeat"]
if len(hashes) >= n and len(set(hashes[-n:])) == 1:
return {"pattern": "repeat", "action_hash": hashes[-1], "count": n}
w = cfg["thrash_window"]
if len(hashes) >= w:
window = hashes[-w:]
distinct = set(window)
if len(distinct) == 2:
# A,B,A,B... : every element differs from its immediate neighbour.
if all(window[i] != window[i + 1] for i in range(len(window) - 1)):
return {"pattern": "thrash", "actions": sorted(distinct), "window": w}
return None
def _detect_no_progress(progresses, cfg):
w = cfg["no_progress_window"]
if len(progresses) < w:
return None
# Mark each step that improved on the running best; STALLED if the last W
# steps contain no improvement at all.
best = None
improved = []
for p in progresses:
if p is None:
improved.append(False)
continue
if best is None or p > best:
improved.append(True)
best = p
else:
improved.append(False)
if not any(improved[-w:]):
return {"window": w, "last_progress": progresses[-1]}
return None
def cmd_verdict(args):
prof = args.profile or lc.profile()
try:
policy = lc.load_policy()
cfg, source = lc.resolve_convergence(policy, prof)
rows = _load_obs(args.run_id)
except lc.PolicyError as exc:
payload = {
"decision": "HALT",
"reason": "fail_closed",
"detail": str(exc),
"run_id": args.run_id,
"provenance": lc.provenance("loop-convergence", None, verified=False),
}
lc.append_audit({"kind": "convergence_halt", **payload})
_emit(payload)
return 3
hashes = [r.get("action_hash") for r in rows]
progresses = [r.get("progress") for r in rows]
osc = _detect_oscillation(hashes, cfg)
stall = None if osc else _detect_no_progress(progresses, cfg)
if osc or stall:
verdict = "OSCILLATING" if osc else "STALLED"
on_stall = cfg.get("on_stall", "escalate")
decision = "HALT" if on_stall == "halt" else "ESCALATE"
payload = {
"decision": decision,
"verdict": verdict,
"run_id": args.run_id,
"profile": prof,
"policy_source": source,
"thresholds": cfg,
"observations": len(rows),
"evidence": osc or stall,
"on_stall": on_stall,
"provenance": lc.provenance("loop-convergence", None, verified=policy is not None),
}
lc.append_audit({"kind": "convergence_" + verdict.lower(), **payload})
_emit(payload)
return 3 if decision == "HALT" else 4
payload = {
"decision": "CONVERGING",
"verdict": "CONVERGING",
"run_id": args.run_id,
"profile": prof,
"policy_source": source,
"thresholds": cfg,
"observations": len(rows),
"provenance": lc.provenance("loop-convergence", None, verified=policy is not None),
}
_emit(payload)
return 0
def build_parser():
p = argparse.ArgumentParser(description="CASAN Convergence Detector (Plan-17 T2)")
sub = p.add_subparsers(dest="cmd", required=True)
o = sub.add_parser("observe", help="Record a per-step observation")
o.add_argument("--run-id", required=True)
o.add_argument("--step", type=int, required=True)
o.add_argument("--action-hash", required=True)
o.add_argument("--progress", type=float, required=True)
o.set_defaults(func=cmd_observe)
v = sub.add_parser("verdict", help="Classify the run: CONVERGING|STALLED|OSCILLATING")
v.add_argument("--run-id", required=True)
v.add_argument("--profile", default=None)
v.set_defaults(func=cmd_verdict)
return p
def main(argv=None):
args = build_parser().parse_args(argv)
try:
return args.func(args)
except lc.PolicyError as exc:
_emit({"decision": "HALT", "reason": "fail_closed", "detail": str(exc)})
return 3
except Exception as exc: # never fail open
_emit({"decision": "HALT", "reason": "internal_error", "detail": str(exc)}, sys.stderr)
return 3
if __name__ == "__main__":
raise SystemExit(main())