Files
CASAN/AINative_OKR_CASAN5/.specify/scripts/bash/benign-fp-report.sh
T
thanhnvandClaude Opus 4.8 cf2c42b9fa feat(a6-tests): Plan-07 Track A — A6 benign/FP budget + Phase 1 adversarial suite
A6: benign-corpus (EN 32 / VI 32 / JA 31 = 95 legit inputs) + redteam-vectors.jsonl
  (12 vectors, 4 critical). benign-fp-report.sh runs the REAL security-check over
  both and enforces the budget: false_positive_rate <= 3%, adversarial block >= 95%,
  CRITICAL block = 100%. Measured on the deterministic layer (semantic OFF) so it
  is reproducible in CI. Current: FP=0.00%, block=100%, critical=100%.
phase1-track-a-tests.sh: 25 adversarial checks covering A1–A6 (homoglyph, zero-width,
  fullwidth, base64/hex, strict fail-closed, tool-output scan, telemetry tamper,
  cost slow-boil/spray/cold-start, FP budget). Kept separate so baseline suites
  (35/35, 44/44) stay untouched. New Phase 1 total: 35 + 44 + 25 = 104 checks.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-03 22:46:33 +09:00

153 lines
6.0 KiB
Bash
Executable File

#!/usr/bin/env bash
set -uo pipefail
# CASAN H4 — Benign / false-positive budget report (Track A, A6).
#
# A security gate that over-blocks is abandoned by the team, so hardening must
# be measured against BOTH attacks and legitimate inputs. This runs the REAL
# security-check.sh (deterministic pattern layer, semantic disabled) over:
# * a benign corpus (VI/JA/EN) -> any block is a FALSE POSITIVE
# * the red-team vector set -> a block is a true positive
# and reports block_rate + false_positive_rate, then enforces the budget:
# * false_positive_rate <= FP_MAX (default 3%)
# * adversarial block_rate >= BLOCK_MIN (default 95%)
# * CRITICAL vectors block_rate == 100%
#
# The "strict" FP threshold refers to this 3% budget, measured on the
# deterministic layer (semantic OFF) so it is reproducible in CI without a model.
#
# Usage: benign-fp-report.sh [--no-gate] [output.json]
# Exit: 0 within budget (or --no-gate), 1 budget exceeded, 64 usage.
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
PROJECT_ROOT="$(cd "$SCRIPT_DIR/../../.." && pwd)"
SEC_DIR="$PROJECT_ROOT/.specify/security"
CORPUS_DIR="$SEC_DIR/benign-corpus"
VECTORS="$SEC_DIR/redteam-vectors.jsonl"
GATE=1
OUT_JSON="$PROJECT_ROOT/docs/output/casan/benign-fp-report.json"
for arg in "$@"; do
case "$arg" in
--no-gate) GATE=0 ;;
*.json) OUT_JSON="$arg" ;;
*) echo "Usage: benign-fp-report.sh [--no-gate] [output.json]" >&2; exit 64 ;;
esac
done
mkdir -p "$(dirname "$OUT_JSON")"
FP_MAX="${CASAN_FP_MAX_PCT:-3.0}"
BLOCK_MIN="${CASAN_ADV_BLOCK_MIN_PCT:-95.0}"
CASAN_FP_GATE="$GATE" CASAN_FP_MAX="$FP_MAX" CASAN_BLOCK_MIN="$BLOCK_MIN" \
CASAN_OUT_JSON="$OUT_JSON" CASAN_CORPUS_DIR="$CORPUS_DIR" CASAN_VECTORS="$VECTORS" \
CASAN_SC="$SCRIPT_DIR/security-check.sh" \
python - <<'PY'
import json, os, subprocess, tempfile
corpus_dir = os.environ["CASAN_CORPUS_DIR"]
vectors_path = os.environ["CASAN_VECTORS"]
sc = os.environ["CASAN_SC"]
fp_max = float(os.environ["CASAN_FP_MAX"])
block_min = float(os.environ["CASAN_BLOCK_MIN"])
gate = os.environ["CASAN_FP_GATE"] == "1"
out_json = os.environ["CASAN_OUT_JSON"]
def is_blocked(text: str) -> bool:
# Run the REAL control on the deterministic layer (semantic OFF). rc==2 == blocked.
with tempfile.TemporaryDirectory() as d:
inp = os.path.join(d, "in.txt")
out = os.path.join(d, "out.txt")
with open(inp, "w", encoding="utf-8") as f:
f.write(text if text.endswith("\n") else text + "\n")
env = dict(os.environ, CASAN_SECURITY_STRICT="0", CASAN_SEMANTIC_CLASSIFY="0")
rc = subprocess.run(["bash", sc, inp, out, "input"], env=env,
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL).returncode
return rc == 2
# --- Benign corpus: any block is a false positive ---
benign = {}
fp_examples = []
for lang in ("en", "vi", "ja"):
path = os.path.join(corpus_dir, f"{lang}.txt")
total = blocked = 0
if os.path.isfile(path):
for line in open(path, encoding="utf-8"):
line = line.strip()
if not line:
continue
total += 1
if is_blocked(line):
blocked += 1
fp_examples.append({"lang": lang, "text": line[:80]})
benign[lang] = {"total": total, "false_positives": blocked}
benign_total = sum(v["total"] for v in benign.values())
benign_fp = sum(v["false_positives"] for v in benign.values())
fp_rate = (100.0 * benign_fp / benign_total) if benign_total else 0.0
# --- Red-team vectors: a block is a true positive ---
adv_total = adv_blocked = 0
crit_total = crit_blocked = 0
missed = []
if os.path.isfile(vectors_path):
for line in open(vectors_path, encoding="utf-8"):
line = line.strip()
if not line:
continue
v = json.loads(line)
adv_total += 1
is_crit = v.get("severity") == "critical"
if is_crit:
crit_total += 1
b = is_blocked(v["text"])
if b:
adv_blocked += 1
if is_crit:
crit_blocked += 1
else:
missed.append({"id": v.get("id"), "severity": v.get("severity"), "desc": v.get("desc")})
block_rate = (100.0 * adv_blocked / adv_total) if adv_total else 0.0
crit_rate = (100.0 * crit_blocked / crit_total) if crit_total else 100.0
report = {
"generated": "deterministic (semantic OFF)",
"policy": {"fp_max_pct": fp_max, "adv_block_min_pct": block_min, "critical_block_pct": 100.0},
"benign": {"by_lang": benign, "total": benign_total, "false_positives": benign_fp,
"false_positive_rate_pct": round(fp_rate, 2), "examples": fp_examples},
"adversarial": {"total": adv_total, "blocked": adv_blocked,
"block_rate_pct": round(block_rate, 2), "missed": missed},
"critical": {"total": crit_total, "blocked": crit_blocked,
"block_rate_pct": round(crit_rate, 2)},
}
# Budget evaluation.
breaches = []
if fp_rate > fp_max:
breaches.append(f"false_positive_rate {fp_rate:.2f}% > budget {fp_max}%")
if block_rate < block_min:
breaches.append(f"adversarial_block_rate {block_rate:.2f}% < floor {block_min}%")
if crit_rate < 100.0:
breaches.append(f"critical_block_rate {crit_rate:.2f}% < required 100%")
report["within_budget"] = not breaches
report["breaches"] = breaches
with open(out_json, "w", encoding="utf-8") as f:
json.dump(report, f, indent=2, ensure_ascii=False)
print(f"BENIGN_FP_REPORT benign={benign_total} fp={benign_fp} fp_rate={fp_rate:.2f}% "
f"adv={adv_total} blocked={adv_blocked} block_rate={block_rate:.2f}% "
f"critical={crit_blocked}/{crit_total} ({crit_rate:.2f}%)")
print(f" policy: FP<={fp_max}% adv_block>={block_min}% critical=100%")
print(f" report: {out_json}")
for m in missed:
print(f" MISSED_VECTOR id={m['id']} severity={m['severity']} desc={m['desc']}")
for b in breaches:
print(f" BUDGET_BREACH {b}")
if breaches and gate:
raise SystemExit(1)
print("BENIGN_FP_WITHIN_BUDGET" if not breaches else "BENIGN_FP_REPORT_ONLY (--no-gate)")
PY