import { spawnSync } from 'node:child_process'; import { createHash } from 'node:crypto'; import { existsSync, mkdirSync, readFileSync, writeFileSync, unlinkSync, appendFileSync } from 'node:fs'; import { tmpdir } from 'node:os'; import { dirname, join } from 'node:path'; import { fileURLToPath } from 'node:url'; // WP-B (H3): model judge gate helpers — AND(rule, model); SKIP is non-blocking. const __filename = fileURLToPath(import.meta.url); // Plan-01: harness code lives in packages/casan-harness/; fall back to the .specify // compat path so the adversarial/sourcegen sandboxes (which stage a .specify/ tree) // keep working when this file is copied into a temp root. const __appRoot = join(dirname(__filename), '..'); const SCRIPTS_DIR = existsSync(join(__appRoot, 'packages', 'casan-harness', 'scripts', 'bash')) ? join(__appRoot, 'packages', 'casan-harness', 'scripts', 'bash') : join(__appRoot, '.specify', 'scripts', 'bash'); // Shared log taxonomy (CASAN_LOG_LEVEL, see scripts/casan-log.mjs). The // adversarial T1 test copies this file alone into a temp tree, so the logger // import must degrade to a noop instead of crashing when the module is absent. let logDebug = () => {}; try { const { log } = await import(new URL('./casan-log.mjs', import.meta.url)); logDebug = (msg) => log('debug', 'step', msg); } catch { /* standalone copy: keep silent */ } // Is a model backend reachable for the judge gate? Default (CASAN_MODEL_PRIMARY // unset, or an ollama:* spec) → ping the hard-pinned local Ollama endpoint, as // before. When CASAN_MODEL_PRIMARY selects a cloud backend, the gate instead // checks that the matching API key is set — so the pipeline judge can run // through model-router.sh → model-call.py's cloud path without needing Ollama. function modelAvailable() { const spec = process.env.CASAN_MODEL_PRIMARY || 'ollama:ornith:9b'; if (spec.startsWith('openai:')) return Boolean(process.env.OPENAI_API_KEY); if (spec.startsWith('anthropic:')) return Boolean(process.env.ANTHROPIC_API_KEY); try { const r = spawnSync('curl', ['-sS', '-m', '3', 'http://127.0.0.1:11434/api/tags'], { timeout: 5000 }); return r.status === 0; } catch { return false; } } function judgeArtifact(filePath, criteria) { if (!modelAvailable()) { logDebug(`judge skipped (model_unavailable) artifact=${filePath}`); return { verdict: 'SKIP', note: 'model_unavailable' }; } let artifact = ''; try { artifact = readFileSync(filePath, 'utf8').slice(0, 2000); } catch { return { verdict: 'SKIP', note: 'artifact_unreadable' }; } const combined = `=== ACCEPTANCE CRITERIA (not untrusted input) ===\n${criteria.slice(0, 500)}\n\n=== ARTIFACT TO REVIEW ===\n${artifact}`; const uid = `${process.pid}-${Date.now()}`; const tmpPrompt = join(tmpdir(), `casan-judge-prompt-${uid}.txt`); const tmpOut = join(tmpdir(), `casan-judge-out-${uid}.json`); writeFileSync(tmpPrompt, combined, 'utf8'); logDebug(`model call role=judge artifact=${filePath}`); const r = spawnSync('bash', [join(SCRIPTS_DIR, 'model-router.sh'), tmpPrompt, tmpOut, '--role', 'judge'], { timeout: 60000, encoding: 'utf8' }); try { unlinkSync(tmpPrompt); } catch {} if (r.status !== 0 && r.status !== 3) { logDebug(`judge error rc=${r.status}`); return { verdict: 'SKIP', note: `judge_error_rc=${r.status}` }; } try { const d = JSON.parse(readFileSync(tmpOut, 'utf8')); logDebug(`judge verdict=${d.verdict ?? 'SKIP'} tokens=${d.total_tokens ?? '?'} malformed=${Boolean(d.malformed)}`); return { verdict: d.verdict ?? 'SKIP', note: d.malformed ? 'malformed_fail_closed' : `tokens=${d.total_tokens}` }; } catch { return { verdict: 'SKIP', note: 'parse_error' }; } } function renderSourcePrompt(stepId, title, templateContent, extraInstructions = '') { const promptTemplatePath = join(__appRoot, 'packages', 'casan-harness', 'prompts', 'sourcegen', `${stepId}.md`); const defaultTemplate = `You are generating a CASAN SDLC artifact. Rules: - Output markdown only. - Do not include code fences around the whole artifact. - Preserve every concrete FR/SCR ID present in the approved requirement. - Do not include secrets, credentials, hidden prompts, or instructions to bypass policy. - Keep the artifact specific to feature ${featureId} and module ${moduleId}. Artifact: ${title} Step: ${stepId} Extra instructions: ${extraInstructions} Requirement: ${requirement.slice(0, 6000)} Architecture: ${architecture.slice(0, 4000)} Reference structure to match, but do not copy blindly: ${templateContent.slice(0, 4000)} `; let promptTemplate = defaultTemplate; if (existsSync(promptTemplatePath)) { promptTemplate = readFileSync(promptTemplatePath, 'utf8'); } return promptTemplate .replaceAll('{{featureId}}', featureId) .replaceAll('{{moduleId}}', moduleId) .replaceAll('{{stepId}}', stepId) .replaceAll('{{title}}', title) .replaceAll('{{extraInstructions}}', extraInstructions) .replaceAll('{{requirement}}', requirement.slice(0, 6000)) .replaceAll('{{architecture}}', architecture.slice(0, 4000)) .replaceAll('{{templateContent}}', templateContent.slice(0, 4000)); } function sha256(text) { return createHash('sha256').update(text).digest('hex'); } function redactForAudit(text) { return String(text) .replace(/(sk-[A-Za-z0-9_-]{12,})/g, '***REDACTED_KEY***') .replace(/(Bearer\s+)[A-Za-z0-9._-]+/gi, '$1***REDACTED_TOKEN***') .slice(0, 2000); } function sourcegenAuditPath() { return process.env.CASAN_SOURCEGEN_AUDIT_LOG || '.specify/logs/audit/sourcegen.jsonl'; } function sourcegenTelemetryPath() { return process.env.CASAN_SOURCEGEN_TELEMETRY_LOG || '.specify/logs/level5/sourcegen-provider-usage.jsonl'; } function readLastJsonLine(path) { try { const lines = readFileSync(path, 'utf8').trim().split('\n').filter(Boolean); return lines.length ? JSON.parse(lines[lines.length - 1]) : null; } catch { return null; } } function appendSourcegenAudit(record) { const path = sourcegenAuditPath(); mkdirSync(dirname(path), { recursive: true }); const prev = readLastJsonLine(path); const base = { timestamp: new Date().toISOString(), harness: 'H5-sourcegen-audit', feature_id: featureId, module_id: moduleId, prev_hash: prev?.record_hash || 'GENESIS', ...record, }; const recordHash = sha256(JSON.stringify(base)); appendFileSync(path, `${JSON.stringify({ ...base, record_hash: recordHash })}\n`, 'utf8'); } function appendSourcegenTelemetry(record) { const path = sourcegenTelemetryPath(); mkdirSync(dirname(path), { recursive: true }); appendFileSync(path, `${JSON.stringify({ timestamp: new Date().toISOString(), harness: 'H6-sourcegen-telemetry', feature_id: featureId, module_id: moduleId, ...record, })}\n`, 'utf8'); } function generateArtifact({ stepId, title, templateContent, required = [], extraInstructions = '' }) { if ((process.env.CASAN_GEN_MODE || 'template') !== 'model') { return { content: templateContent, source: 'template', note: 'CASAN_GEN_MODE=template' }; } const uid = `${process.pid}-${Date.now()}-${stepId}`; const tmpPrompt = join(tmpdir(), `casan-generate-prompt-${uid}.txt`); const tmpOut = join(tmpdir(), `casan-generate-out-${uid}.json`); const tmpDraft = join(tmpdir(), `casan-generate-draft-${uid}.md`); try { const prompt = renderSourcePrompt(stepId, title, templateContent, extraInstructions); const promptHash = sha256(prompt); writeFileSync(tmpPrompt, prompt, 'utf8'); logDebug(`model call role=generate step=${stepId}`); const r = spawnSync( 'bash', [join(SCRIPTS_DIR, 'model-router.sh'), tmpPrompt, tmpOut, '--role', 'generate'], { timeout: Number(process.env.CASAN_SOURCEGEN_MODEL_TIMEOUT_MS || 300000), encoding: 'utf8', env: { ...process.env, CASAN_STEP_NAME: stepId, CASAN_MODEL_TIMEOUT_SEC: process.env.CASAN_MODEL_TIMEOUT_SEC || process.env.CASAN_SOURCEGEN_MODEL_TIMEOUT_SEC || '240', }, }, ); if (r.status !== 0) { appendSourcegenAudit({ step_id: stepId, event: 'generate_fallback', reason: `model_generate_rc=${r.status}`, prompt_sha256: promptHash, prompt_text: redactForAudit(prompt), source: 'template-fallback', }); return { content: templateContent, source: 'template-fallback', note: `model_generate_rc=${r.status}` }; } const d = JSON.parse(readFileSync(tmpOut, 'utf8')); const generated = String(d.text || '').trim(); if (generated.length < 80) { appendSourcegenAudit({ step_id: stepId, event: 'generate_fallback', reason: 'model_output_too_short', prompt_sha256: promptHash, prompt_text: redactForAudit(prompt), model_id: d.model_id, route: d.route, input_tokens: d.input_tokens, output_tokens: d.output_tokens, total_tokens: d.total_tokens, source: 'template-fallback', }); return { content: templateContent, source: 'template-fallback', note: 'model_output_too_short' }; } const missing = required.filter((token) => !generated.includes(token)); if (missing.length > 0) { appendSourcegenAudit({ step_id: stepId, event: 'generate_fallback', reason: `missing_required=${missing.join(',')}`, prompt_sha256: promptHash, prompt_text: redactForAudit(prompt), model_id: d.model_id, route: d.route, input_tokens: d.input_tokens, output_tokens: d.output_tokens, total_tokens: d.total_tokens, source: 'template-fallback', }); return { content: templateContent, source: 'template-fallback', note: `missing_required=${missing.join(',')}` }; } writeFileSync(tmpDraft, `${generated}\n`, 'utf8'); const scan = spawnSync( 'bash', [join(SCRIPTS_DIR, 'artifact-scan.sh'), tmpDraft, `sourcegen-${stepId}`], { timeout: 30000, encoding: 'utf8' }, ); if (scan.status !== 0) { appendSourcegenAudit({ step_id: stepId, event: 'generate_fallback', reason: `artifact_scan_rc=${scan.status}`, prompt_sha256: promptHash, prompt_text: redactForAudit(prompt), model_id: d.model_id, route: d.route, input_tokens: d.input_tokens, output_tokens: d.output_tokens, total_tokens: d.total_tokens, source: 'template-fallback', }); return { content: templateContent, source: 'template-fallback', note: `artifact_scan_rc=${scan.status}` }; } appendSourcegenAudit({ step_id: stepId, event: 'generate_accept', prompt_sha256: promptHash, prompt_text: redactForAudit(prompt), output_sha256: sha256(generated), model_id: d.model_id, route: d.route, input_tokens: d.input_tokens, output_tokens: d.output_tokens, total_tokens: d.total_tokens, source: 'model', }); appendSourcegenTelemetry({ provider: String(d.route || '').split(':')[0] || 'unknown', model: d.model_id || 'unknown', step: stepId, role: 'generate', input_tokens: Number(d.input_tokens || 0), output_tokens: Number(d.output_tokens || 0), total_tokens: Number(d.total_tokens || 0), status: 'success', }); return { content: `${generated}\n`, source: 'model', note: `tokens=${d.total_tokens ?? '?'}` }; } catch (err) { return { content: templateContent, source: 'template-fallback', note: `generate_error=${err?.name || 'Error'}` }; } finally { for (const path of [tmpPrompt, tmpOut, tmpDraft]) { try { unlinkSync(path); } catch {} } } } // T1: Real rollback wiring — checkpoint before overwrite, restore on REJECTED verdict function checkpointArtifact(filePath) { try { readFileSync(filePath, 'utf8'); } catch { return null; } // file absent → nothing to checkpoint const r = spawnSync('bash', [join(SCRIPTS_DIR, 'rollback-manager.sh'), 'checkpoint', filePath], { encoding: 'utf8', timeout: 10000 }); if (r.status !== 0) return null; const m = (r.stdout || '').match(/transaction_id=(\S+)/); logDebug(`checkpoint artifact=${filePath} tx=${m ? m[1] : 'none'}`); return m ? m[1] : null; } function executeRollback(txId) { if (!txId) return false; const r = spawnSync('bash', [join(SCRIPTS_DIR, 'rollback-manager.sh'), 'execute', txId], { encoding: 'utf8', timeout: 10000 }); logDebug(`rollback execute tx=${txId} rc=${r.status}`); return r.status === 0; } const step = process.argv[2]; const attempt = process.argv[3] ?? '1'; const outputPath = process.env.CASAN_OUTPUT; let projectTools; try { projectTools = await import(new URL('./casan-project.mjs', import.meta.url)); } catch { // Hermetic legacy tests copy this file alone. Production always ships the loader. projectTools = null; } const legacyProject = { project_id: 'okr', display_name: 'OKR Web Application', domain_root: 'apps/okr/domain', requirements: existsSync('apps/okr/domain/input/okr-requirement.md') ? 'apps/okr/domain/input/okr-requirement.md' : 'docs/input/okr-requirement.md', architecture: 'docs/technical_architecture.md', feature: { id: '001-okr-web-app', module_id: 'MOD-01', slug: 'okr-management', title: 'OKR Web Application' }, source_roots: ['apps/okr/backend', 'apps/okr/frontend'], commands: { build: [], test: [] }, verification: [], artifacts_root: 'docs/output', implementation_evidence: [ 'apps/okr/backend/src/auth/auth.service.ts', 'apps/okr/backend/src/objectives/objectives.service.ts', 'apps/okr/backend/src/key-results/key-results.service.ts', 'apps/okr/backend/test/e2e.test.ts', 'apps/okr/backend/test/services.test.ts', 'apps/okr/frontend/src/lib/api.ts', 'apps/okr/backend/prisma/seed.ts', ], tech_stack: 'NestJS, Prisma Client, SQLite, React, Vite, Tailwind CSS, Zod, TanStack Query', quality: { minimum_requirements: 1, required_spec_sections: ['Requirements', 'Acceptance Criteria', 'Input Validation Rules', 'Source Trace'], required_plan_sections: ['Architecture', 'Implementation Workstreams', 'Tests', 'Golden regression test', 'Rollback strategy'], }, manifest_path: 'legacy', }; const project = projectTools ? projectTools.loadProjectManifest({ root: __appRoot }) : legacyProject; const isLegacy = project.manifest_path === 'legacy'; const featureId = project.feature.id; const moduleId = project.feature.module_id; if (!outputPath) { throw new Error('CASAN_OUTPUT is required'); } logDebug(`step=${step} attempt=${attempt} output=${outputPath}`); const artifactPaths = projectTools ? projectTools.projectArtifactPaths(project) : { logDir: `docs/output/output_logs/${featureId}`, reportsDir: `docs/output/output_logs/${featureId}/reports`, specsDir: `docs/output/specs/${featureId}`, srsPath: 'docs/output/ipa-docs/srs/srs-mod01-okr-management.md', bdPath: 'docs/output/ipa-docs/bd/bd-mod01-okr-management.md', ddPath: 'docs/output/ipa-docs/dd/dd-mod01-okr-management.md', testcasePath: 'docs/output/ipa-docs/testcase/testcase-mod01-okr-management.md', }; const dirs = [ artifactPaths.reportsDir, dirname(artifactPaths.srsPath), dirname(artifactPaths.bdPath), dirname(artifactPaths.ddPath), dirname(artifactPaths.testcasePath), `${artifactPaths.specsDir}/contracts`, ]; dirs.forEach((dir) => mkdirSync(dir, { recursive: true })); const requirement = readFileSync(project.requirements, 'utf8'); const architecture = readFileSync(project.architecture, 'utf8'); const requirementIds = projectTools ? projectTools.extractIds(requirement, 'FR') : [...new Set(requirement.match(/\bFR-\d{2,}\b/g) ?? [])]; const screenIds = projectTools ? projectTools.extractIds(requirement, 'SCR') : [...new Set(requirement.match(/\bSCR-\d{2,}\b/g) ?? [])]; const requirementDetails = projectTools ? projectTools.requirementLines(requirement, requirementIds) : requirementIds.map((id) => (requirement.split(/\r?\n/).find((line) => line.includes(id)) || id).replace(/^\s*[-|#*]+\s*/, '').trim()); const screenDetails = projectTools ? projectTools.requirementLines(requirement, screenIds) : screenIds.map((id) => (requirement.split(/\r?\n/).find((line) => line.includes(id)) || id).replace(/^\s*[-|#*]+\s*/, '').trim()); if (requirementIds.length < Number(project.quality.minimum_requirements || 1)) { throw new Error(`CASAN_REQUIREMENTS_INSUFFICIENT: found=${requirementIds.length}`); } function write(path, content) { mkdirSync(dirname(path), { recursive: true }); writeFileSync(path, content, 'utf8'); return path; } function resultBlock({ status = 'COMPLETE', verdict = 'APPROVED', artifacts = [], issues = [] }) { return `\n\n`; } function report(path, title, body, verdict = 'APPROVED', artifacts = [], issues = []) { finalTitle = title; finalBody = body; finalVerdict = verdict; finalArtifacts = artifacts; finalIssues = issues; write(path, `${title}\n\n${body}\n${resultBlock({ verdict, artifacts, issues })}`); } const srsPath = artifactPaths.srsPath; const bdPath = artifactPaths.bdPath; const specPath = `${artifactPaths.specsDir}/spec.md`; const planPath = `${artifactPaths.specsDir}/plan.md`; const dataModelPath = `${artifactPaths.specsDir}/data-model.md`; const researchPath = `${artifactPaths.specsDir}/research.md`; const quickstartPath = `${artifactPaths.specsDir}/quickstart.md`; const contractPath = `${artifactPaths.specsDir}/contracts/openapi.md`; const ddPath = artifactPaths.ddPath; const testcasePath = artifactPaths.testcasePath; const tasksPath = `${artifactPaths.specsDir}/tasks.md`; const implementationPath = `${artifactPaths.specsDir}/implementation.md`; const implementationAcceptancePath = `${artifactPaths.specsDir}/implementation.accepted.json`; const testRunPath = `${artifactPaths.reportsDir}/12-run-tests-report.md`; const codeReviewPath = `${artifactPaths.reportsDir}/11-review-code-report.md`; let finalTitle = `# ${step}`; let finalBody = ''; let finalVerdict = 'APPROVED'; let finalArtifacts = []; let finalIssues = []; // Sidecar stores the checkpoint tx-id between step 05-plan and step 06-reviewplan const CHECKPOINT_SIDECAR = `${artifactPaths.specsDir}/plan.checkpoint.txid`; const implementationFiles = project.implementation_evidence || []; const reportPath = (name) => `${artifactPaths.reportsDir}/${name}`; const bullets = (items) => items.map((item) => `- ${item}`).join('\n'); const commandsMarkdown = (commands) => commands.map((command) => `- \`${command.join(' ')}\``).join('\n') || '- none'; function runArgvCommands(kind) { const commands = project.commands[kind] || []; const records = []; for (const command of commands) { const result = spawnSync(command[0], command.slice(1), { cwd: __appRoot, encoding: 'utf8', timeout: Number(process.env.CASAN_SOURCEGEN_TEST_TIMEOUT_MS || 300000), env: { ...process.env }, }); records.push({ command, status: result.status, stdout: result.stdout || '', stderr: result.stderr || '' }); if (result.status !== 0) break; } return records; } switch (step) { case '01-srs': { const templateContent = `# SRS — ${project.feature.title}\n\n## 1. Purpose\nDeliver ${project.display_name} from the approved requirement and architecture without inventing product behavior.\n\n## 2. Scope\nFeature ${featureId}; module ${moduleId}; project ${project.project_id}.\n\n## 3. Functional Requirements\n${bullets(requirementDetails)}\n\n## 4. Non Functional Requirements\nArchitecture constraints are authoritative. Security, validation, traceability, build, test, rollback, and evidence gates are mandatory.\n\n## Metrics\nFunctional requirements extracted: ${requirementIds.length}. Screens extracted: ${screenIds.length}.\n`; const generated = generateArtifact({ stepId: '01-srs', title: 'Software Requirements Specification', templateContent, required: requirementIds, extraInstructions: 'Create an SRS with purpose, scope, functional requirements, non-functional requirements, and traceable FR IDs.', }); write(srsPath, generated.content); report(reportPath('01-srs-report.md'), '# STEP 1: SRS Generation Report', `Generated ${srsPath} from ${project.requirements}. source=${generated.source} note=${generated.note}.`, 'APPROVED', [srsPath]); break; } case '02-bd': { const templateContent = `# Business Design — ${project.feature.title}\n\n## Screen Layout\n${screenDetails.length ? bullets(screenDetails) : '- No screen identifiers are declared; preserve the approved service boundary.'}\n\n## User Journeys\nEach functional requirement must have a normal flow, validation failure, authorization failure where applicable, and observable completion state.\n\n## API Boundary\nNetwork and data boundaries must follow ${project.architecture}. UI components may not bypass the approved API client boundary.\n`; const generated = generateArtifact({ stepId: '02-bd', title: 'Business Design', templateContent, required: [...screenIds, 'API Boundary'], extraInstructions: 'Create a business design with screen layout for SCR-00 through SCR-04 and an API Boundary section.', }); write(bdPath, generated.content); report(reportPath('02-bd-report.md'), '# STEP 2: Business Design Report', `Generated ${bdPath}. source=${generated.source} note=${generated.note}.`, 'APPROVED', [bdPath]); break; } case '03-spec': { const templateContent = `# Feature Specification: ${project.feature.title}\n\n## Feature ID\n${featureId}\n\n## Requirements\n${bullets(requirementDetails)}\n\n## Acceptance Criteria\n${requirementIds.map((id) => `- ${id}: normal behavior, boundary validation, authorization where applicable, persistence/effect, and failure response are testable.`).join('\n')}\n- Golden output drift fails the quality gate.\n\n## Input Validation Rules\n- Reject missing, malformed, out-of-range, and unauthorized inputs before side effects.\n- Do not expose secrets or internal error details.\n\n## Source Trace\nRequirement: ${project.requirements}. Architecture: ${project.architecture}. Manifest: ${project.manifest_path}.\n`; const generated = generateArtifact({ stepId: '03-spec', title: 'Feature Specification', templateContent, required: isLegacy ? [...requirementIds, 'Acceptance Criteria'] : [...requirementIds, ...project.quality.required_spec_sections], extraInstructions: 'Create a feature specification with FR coverage, role filtering, validation rules, acceptance criteria, and source trace.', }); write(specPath, generated.content); report(reportPath('03-spec-report.md'), '# STEP 3: Specify Report', `Generated ${specPath}. source=${generated.source} note=${generated.note}.`, 'APPROVED', [specPath]); break; } case '04-reviewspec': { const spec = readFileSync(specPath, 'utf8'); const required = isLegacy ? requirementIds : [...requirementIds, ...project.quality.required_spec_sections]; const missing = required.filter((id) => !spec.includes(id)); let verdict = missing.length === 0 ? 'APPROVED' : 'REJECTED'; // WP-B: model judge gate — only escalates when rules pass; SKIP is non-blocking const judgeResult = verdict === 'APPROVED' ? judgeArtifact(specPath, `Spec must cover every requirement (${requirementIds.join(', ')}), concrete acceptance criteria, validation rules, and source trace without inventing behavior.`) : { verdict: 'SKIP', note: 'rule_already_rejected' }; if (judgeResult.verdict === 'REJECTED') verdict = 'REJECTED'; const judgeNote = `model-judge: ${judgeResult.verdict} (${judgeResult.note})`; report( reportPath('04-review-spec-report.md'), '## Spec Conformance Review Report', `Criteria checked: requirement coverage, validation, acceptance, source trace, golden regression. Missing: ${missing.join(', ') || 'none'}. ${judgeNote}.`, verdict, [specPath], missing, ); break; } case '05-plan': { const incomplete = attempt === '1'; // T1: checkpoint existing plan before overwriting so rollback-manager.sh execute can restore it if (attempt !== '1') { const txId = checkpointArtifact(planPath); if (txId) { try { writeFileSync(CHECKPOINT_SIDECAR, txId, 'utf8'); } catch {} } } const architecturePlan = isLegacy ? `## Backend Modules\n- auth, users, objectives, and key-results follow the approved NestJS module boundaries.` : `## Architecture\n${project.tech_stack || 'See architecture input.'}\nAuthoritative document: ${project.architecture}.\n\n## Implementation Workstreams\n${project.source_roots.map((root) => `- Implement traced requirements under ${root}.`).join('\n')}`; const templateContent = `# Implementation Plan: ${project.feature.title}\n\n${architecturePlan}\n\n## Tests\n${commandsMarkdown(project.commands.test)}\n${incomplete ? '- TODO: define golden regression and rollback strategy.\n' : '- Golden regression test compares deterministic artifacts with the project Domain Pack golden-runs.\n- Rollback strategy checkpoints changed artifacts and reverses an approved patch when verification fails.\n'}\n## Build\n${commandsMarkdown(project.commands.build)}\n`; const generated = generateArtifact({ stepId: '05-plan', title: 'Implementation Plan', templateContent, required: isLegacy ? ['Backend Modules', 'Tests', ...(incomplete ? [] : ['Golden regression test', 'Rollback strategy'])] : ['Architecture', 'Implementation Workstreams', 'Tests', ...(incomplete ? [] : ['Golden regression test', 'Rollback strategy'])], extraInstructions: incomplete ? 'Create the first implementation plan draft. Leave companion artifacts for the retry loop so review-plan can reject missing artifacts.' : 'Create the final implementation plan with Backend Modules, Tests, Golden regression test, Rollback strategy, and build commands.', }); write(planPath, generated.content); if (!incomplete) { write(dataModelPath, `# Data Model\n\nDerived from ${project.requirements}. Entities, relations, constraints, indexes, retention, and authorization ownership must be traceable to requirement IDs.\n`); write(researchPath, `# Research\n\nApproved stack: ${project.tech_stack || 'see architecture'}. Alternatives may not override ${project.architecture}.\n`); write(quickstartPath, `# Quickstart\n\n## Build\n${commandsMarkdown(project.commands.build)}\n\n## Test\n${commandsMarkdown(project.commands.test)}\n`); write(contractPath, `# API Contract\n\nContract boundaries are derived from ${project.requirements}. No endpoint is accepted without a requirement ID, validation contract, authorization rule, and error response.\n`); } report(reportPath(`05-plan-report-attempt-${attempt}.md`), `# STEP 5: Plan Report Attempt ${attempt}`, `Generated ${planPath}. source=${generated.source} note=${generated.note}.`, 'APPROVED', [planPath]); break; } case '06-reviewplan': { const plan = readFileSync(planPath, 'utf8'); const required = isLegacy ? ['Golden regression test', 'Rollback strategy', 'Backend Modules', 'Tests'] : project.quality.required_plan_sections; const missing = required.filter((phrase) => !plan.includes(phrase)); const artifactMissing = [dataModelPath, researchPath, quickstartPath, contractPath].filter((path) => { try { readFileSync(path, 'utf8'); return false; } catch { return true; } }); const issues = [...missing.map((item) => `missing plan criterion: ${item}`), ...artifactMissing.map((item) => `missing artifact: ${item}`)]; let verdict = issues.length === 0 ? 'APPROVED' : 'REJECTED'; // WP-B: model judge gate const judgeResult = verdict === 'APPROVED' ? judgeArtifact(planPath, `Plan must include ${required.join(', ')} and all companion artifacts. Commands must match the project manifest.`) : { verdict: 'SKIP', note: 'rule_already_rejected' }; if (judgeResult.verdict === 'REJECTED') verdict = 'REJECTED'; const judgeNote = `model-judge: ${judgeResult.verdict} (${judgeResult.note})`; // T1: on REJECTED, execute rollback to restore the previously checkpointed plan let rollbackNote = ''; if (verdict === 'REJECTED') { let txId = null; try { txId = readFileSync(CHECKPOINT_SIDECAR, 'utf8').trim(); } catch {} if (txId) { const restored = executeRollback(txId); rollbackNote = restored ? ` Rollback executed: plan restored to pre-overwrite state (tx=${txId}).` : ` Rollback attempted but failed (tx=${txId}).`; } } report( reportPath(`06-review-plan-report-attempt-${attempt}.md`), `## Plan Conformance Review Report — Attempt ${attempt}`, `Criteria checked against plan.md and required companion artifacts. ${judgeNote}. Verdict is ${verdict}.${rollbackNote}`, verdict, [planPath, dataModelPath, researchPath, quickstartPath, contractPath], issues, ); break; } case '07-dd': { const templateContent = `# Detail Design — ${project.feature.title}\n\n## Component Design\n${project.source_roots.map((root) => `- ${root}: owns only responsibilities assigned by the approved architecture.`).join('\n')}\n\n## Authorization and Validation\nEvery entry point validates input, authenticates identity where required, authorizes the resource, and emits safe errors before side effects.\n\n## Data and Transaction Design\nPersistence, transaction boundaries, concurrency, idempotency, and failure recovery must trace to the data model and requirement IDs.\n\n## Observability\nRecord correlation IDs, safe structured events, latency, failures, and governance evidence without secrets or sensitive payloads.\n`; const generated = generateArtifact({ stepId: '07-dd', title: 'Detail Design', templateContent, required: isLegacy ? ['Backend Design', 'Authorization', 'Progress Calculation'] : ['Component Design', 'Authorization and Validation', 'Data and Transaction Design', 'Observability'], extraInstructions: 'Create a detailed design for backend modules, authorization, database access, and progress calculation.', }); write(ddPath, generated.content); report(reportPath('07-dd-report.md'), '# STEP 7: Detail Design Report', `Generated ${ddPath}. source=${generated.source} note=${generated.note}.`, 'APPROVED', [ddPath]); break; } case '08-testkit': { const testCases = requirementIds.map((id, index) => `- TC-${String(index + 1).padStart(2, '0')} ${id}: normal, boundary, invalid, unauthorized, and persistence/effect assertions.`); testCases.push(`- TC-${String(testCases.length + 1).padStart(2, '0')} Golden regression fails on approved-output drift.`); const templateContent = `# Test Cases — ${moduleId}\n\n${testCases.join('\n')}\n`; const generated = generateArtifact({ stepId: '08-testkit', title: 'Test Case Design', templateContent, required: isLegacy ? ['TC-01', 'TC-02', 'TC-03', 'TC-04', 'TC-05', 'TC-06'] : ['TC-01', ...requirementIds], extraInstructions: 'Create test cases with TC IDs covering login, role-filtering, validation, progress update, and golden drift.', }); write(testcasePath, generated.content); report(reportPath('08-testkit-report.md'), '# STEP 8: Testkit Report', `Generated ${testcasePath}. source=${generated.source} note=${generated.note}.`, 'APPROVED', [testcasePath]); break; } case '09-tasks': { const templateContent = `# Tasks\n\n${requirementIds.map((id) => `- [ ] Implement and test ${id} in its owning source root.`).join('\n')}\n- [ ] Validate all manifest build commands.\n- [ ] Validate all manifest test commands.\n- [ ] Capture traceability, golden regression, security, rollback, and evidence reports.\n`; const generated = generateArtifact({ stepId: '09-tasks', title: 'Implementation Tasks', templateContent, required: isLegacy ? ['Backend', 'Frontend', 'tests', 'Golden'] : [...requirementIds, 'build', 'test', 'golden'], extraInstructions: 'Create a task list that keeps test work and implementation work traceable to the plan and acceptance criteria.', }); write(tasksPath, generated.content); report(reportPath('09-tasks-report.md'), '# STEP 9: Tasks Report', `Generated ${tasksPath}. source=${generated.source} note=${generated.note}.`, 'APPROVED', [tasksPath]); break; } case '10-implement': { const templateContent = `# Implementation Draft: ${project.feature.title}\n\n## Scope\nThe implementation candidate is bounded by the manifest source roots and is not accepted before STEP12 build/test PASS.\n\n## Source Roots\n${bullets(project.source_roots)}\n\n## Implementation Evidence\n${bullets(implementationFiles)}\n\n## Acceptance Gate\nSTEP12 must execute manifest argv commands and write ${implementationAcceptancePath}; no shell interpolation is permitted.\n`; const generated = generateArtifact({ stepId: '10-implement', title: 'Implementation Draft', templateContent, required: isLegacy ? ['Backend Source', 'Frontend Source', 'Acceptance Gate', 'STEP12'] : ['Source Roots', 'Implementation Evidence', 'Acceptance Gate', 'STEP12'], extraInstructions: 'Create a code implementation draft/manifest. Reference concrete source files and state that acceptance requires STEP12 test PASS.', }); write(implementationPath, generated.content); try { unlinkSync(implementationAcceptancePath); } catch {} report( reportPath('10-implement-report.md'), '# STEP 10: Implementation Draft Report', `Generated ${implementationPath}. source=${generated.source} note=${generated.note}. acceptance=pending STEP12.`, 'APPROVED', [implementationPath], ); break; } case '10-reviewcode': case '11-reviewcode': { const missingFiles = implementationFiles.filter((path) => { try { readFileSync(path, 'utf8'); return false; } catch { return true; } }); const issues = missingFiles.map((path) => `missing file: ${path}`); for (const sourceRoot of project.source_roots) { const prefix = `${sourceRoot.replace(/\/$/, '')}/`; if (!project.verification.some((rule) => prefix.startsWith(rule.path_prefix) || rule.path_prefix.startsWith(prefix))) { issues.push(`unmapped verification source root: ${sourceRoot}`); } } let verdict = issues.length === 0 ? 'APPROVED' : 'REJECTED'; // WP-B: model judge gate const judgeTarget = implementationFiles.find((p) => { try { readFileSync(p, 'utf8'); return true; } catch { return false; } }) ?? codeReviewPath; const judgeResult = verdict === 'APPROVED' ? judgeArtifact(judgeTarget, 'Code must follow the approved architecture, validate inputs, enforce authorization where required, avoid hard-coded product fixtures, and have manifest-bound build/test evidence.') : { verdict: 'SKIP', note: 'rule_already_rejected' }; if (judgeResult.verdict === 'REJECTED') verdict = 'REJECTED'; const judgeNote = `model-judge: ${judgeResult.verdict} (${judgeResult.note})`; report( codeReviewPath, '# STEP 10: Code Review Report', `Reviewed ${implementationFiles.length} manifest implementation evidence files across ${project.source_roots.length} source roots. ${judgeNote}.`, verdict, implementationFiles, issues, ); break; } case '12-runtests': { let implementation = ''; try { implementation = readFileSync(implementationPath, 'utf8'); } catch {} let records; if (project.manifest_path === 'legacy' && process.env.CASAN_SOURCEGEN_TEST_CMD) { const legacy = spawnSync('bash', ['-lc', process.env.CASAN_SOURCEGEN_TEST_CMD], { cwd: __appRoot, encoding: 'utf8', timeout: 180000 }); records = [{ command: ['legacy-test-command'], status: legacy.status, stdout: legacy.stdout || '', stderr: legacy.stderr || '' }]; } else { const buildRecords = runArgvCommands('build'); records = buildRecords.some((record) => record.status !== 0) ? buildRecords : [...buildRecords, ...runArgvCommands('test')]; } const failed = records.find((record) => record.status !== 0); const passed = records.length > 0 && !failed; const output = records.map((record) => `$ ${record.command.join(' ')}\n${record.stdout}\n${record.stderr}`).join('\n').trim().slice(0, 12000); const acceptance = { timestamp: new Date().toISOString(), feature_id: featureId, implementation_artifact: implementationPath, implementation_sha256: sha256(implementation), commands: records.map((record) => record.command), exit_codes: records.map((record) => record.status), accepted: passed, }; if (passed) { write(implementationAcceptancePath, `${JSON.stringify(acceptance, null, 2)}\n`); } else { try { unlinkSync(implementationAcceptancePath); } catch {} } report( testRunPath, '# STEP 12: Run Tests Report', `Commands: ${records.map((record) => record.command.join(' ')).join(' ; ')}\nExit codes: ${records.map((record) => record.status).join(', ')}\nAccepted implementation: ${passed ? 'yes' : 'no'}\n\n## Output\n${output || '(no output)'}`, passed ? 'PASS' : 'FAIL', passed ? [implementationPath, implementationAcceptancePath] : [implementationPath], passed ? [] : [`build/test command failed rc=${failed?.status ?? 'no-commands'}`], ); break; } default: throw new Error(`Unknown step: ${step}`); } writeFileSync( outputPath, `${finalTitle}\n\n${finalBody}\n\nGenerated by ${step} attempt ${attempt} for ${featureId}.\n${resultBlock({ verdict: finalVerdict, artifacts: finalArtifacts, issues: finalIssues, })}`, 'utf8', );