/**
 * gates/h65-confidence-gate.ts — H65v2 Confidence-Based Knowledge Freshness
 *
 * PARADIGM SHIFT: From "block until verified" to "assess confidence, retrieve adaptively"
 *
 * Architecture (Generate-Verify-Repair loop):
 *   Phase 1: PRE-GENERATION — Lightweight confidence assessment (no blocking)
 *     - Domain detection (reuses detectVolatileDomains)
 *     - Confidence scoring via 5 signals
 *     - Decision: retrieve-first | generate-first | skip
 *
 *   Phase 2: POST-GENERATION — Verify output correctness
 *     - AST/import validation (does the code reference real modules?)
 *     - Type-check probe (tsc --noEmit on generated file)
 *     - Package.json cross-reference (are deps declared?)
 *     - API signature spot-check (optional web verification)
 *
 *   Phase 3: REPAIR — If verification fails, trigger retrieval + regenerate
 *     - Fetch current docs for failed domains
 *     - Inject fresh context
 *     - Request worker to regenerate failed section
 *
 * Research basis (22 sources):
 *   - Kadavath et al. 2022 "LLMs Know What They Know" — P(IK) self-evaluation
 *   - Kuhn et al. 2023 "Semantic Entropy" — uncertainty via sampling consistency
 *   - Manakul et al. 2023 "SelfCheckGPT" — zero-resource hallucination detection
 *   - Abbasi Yadkori et al. 2024 "Conformal Abstention" — rigorous hallucination bounds
 *   - Wu et al. 2024 "ClashEval" — token probs predict knowledge adoption
 *   - Brown et al. 2024 "Large Language Monkeys" — repeated sampling + auto-verify
 *   - Huang et al. 2023 "LLMs Cannot Self-Correct" — external feedback required
 *   - Shinn et al. 2023 "Reflexion" — verbal reinforcement with test feedback
 *   - Asai et al. 2023 "Self-RAG" — learned retrieval/critique decisions
 *   - Jiang et al. 2023 "FLARE" — forward-looking confidence-based retrieval
 *   - Jeong et al. 2024 "Adaptive-RAG" — complexity-based routing
 *   - CRAG (Meta 2024) — corrective RAG with knowledge refinement
 *   - Peng 2026 "HaS" — speculative retrieval with validation
 *   - Wang 2024/2026 "CARROT" — cost-constrained retrieval MCTS
 *   - Vu et al. 2023 "FreshLLMs" — temporal freshness benchmark
 *   - Olausson et al. 2023 "Self-Repair Silver Bullet?" — repair needs signals
 *   - Ehtesham et al. 2025 "Agentic RAG Survey" — taxonomy of patterns
 *   - Knowledge Conflicts Survey 2024 — parametric vs contextual resolution
 *   - Schick et al. 2023 "Toolformer" — self-taught tool use decisions
 *   - LeDex 2024 — self-debugging with explanations
 *   - Efficient Bayesian Semantic Entropy 2026 — adaptive estimation
 *   - Rowen 2024 — adaptive RAG for hallucination mitigation
 *
 * Key insight: The old H65 gate asked "does this TASK mention a volatile domain?"
 * The new gate asks "does this AGENT have sufficient confidence for this specific API usage?"
 * This shifts from task-level keyword matching to output-level verification.
 */

import { existsSync, readFileSync } from 'node:fs';
import { join, extname } from 'node:path';
import { homedir } from 'node:os';
import { detectVolatileDomains, loadVolatileDomains, type DetectionResult } from './h65-knowledge-freshness-gate.ts';
const { heliosPath } = require('../../lib/helios-root');

// ── Configuration ─────────────────────────────────────────────────────

export interface H65v2Config {
  /** Enable pre-generation confidence check (default: true) */
  preCheck: boolean;
  /** Enable post-generation verification (default: true) */
  postVerify: boolean;
  /** Confidence threshold below which to retrieve-first (0-1, default: 0.4) */
  retrieveFirstThreshold: number;
  /** Confidence threshold above which to skip verification (0-1, default: 0.85) */
  skipVerifyThreshold: number;
  /** Max repair attempts before failing (default: 2) */
  maxRepairAttempts: number;
  /** Enable TypeScript type-check verification (default: true) */
  typeCheckVerify: boolean;
  /** Enable import/package verification (default: true) */
  importVerify: boolean;
  /** Enable API signature web verification (default: false — expensive) */
  apiSignatureVerify: boolean;
  /** Thompson sampling decay rate for confidence learning (default: 0.95) */
  thompsonDecay: number;
}

const DEFAULT_CONFIG: H65v2Config = {
  preCheck: true,
  postVerify: true,
  retrieveFirstThreshold: 0.4,
  skipVerifyThreshold: 0.85,
  maxRepairAttempts: 2,
  typeCheckVerify: true,
  importVerify: true,
  apiSignatureVerify: false,
  thompsonDecay: 0.95,
};

// ── Confidence Signals ────────────────────────────────────────────────

/**
 * Five independent signals that contribute to confidence scoring.
 * Each signal produces a score between 0 (no confidence) and 1 (full confidence).
 *
 * Based on "LLMs Know What They Know" (Kadavath 2022):
 * "Models perform well at predicting P(IK) and partially generalize across tasks"
 */
export interface ConfidenceSignals {
  /** Signal 1: Domain recency — how old is the API relative to training cutoff? */
  domainRecency: number;
  /** Signal 2: Specificity — is the task about a specific API endpoint or general patterns? */
  specificity: number;
  /** Signal 3: Cache freshness — do we have verified recent data for this domain? */
  cacheFreshness: number;
  /** Signal 4: Historical success — Thompson sampling of past outcomes for this domain */
  historicalSuccess: number;
  /** Signal 5: Complexity — single API call vs multi-step integration? */
  complexity: number;
}

/**
 * Compute composite confidence from signals.
 * Weighted geometric mean — any single very-low signal pulls overall confidence down.
 *
 * Research: ClashEval (Wu 2024) shows "the less confident a model is in its initial
 * response, the more likely it is to adopt information in retrieved content."
 * This means low-confidence = high value from retrieval.
 */
export function computeConfidence(signals: ConfidenceSignals): number {
  const weights = {
    domainRecency: 0.30,   // Most important: how old is the API?
    historicalSuccess: 0.25, // Learned from past outcomes
    cacheFreshness: 0.20,  // Do we have verified data?
    specificity: 0.15,     // Generic patterns vs specific APIs
    complexity: 0.10,      // Simple vs complex integration
  };

  // Weighted geometric mean (multiplicative — any zero tanks the score)
  let logSum = 0;
  let totalWeight = 0;
  for (const [key, weight] of Object.entries(weights)) {
    const value = signals[key as keyof ConfidenceSignals];
    if (value <= 0) return 0; // Hard floor: any zero signal = no confidence
    logSum += weight * Math.log(value);
    totalWeight += weight;
  }

  return Math.exp(logSum / totalWeight);
}

// ── Signal Computation ────────────────────────────────────────────────

const TRAINING_CUTOFF = new Date('2024-04-01'); // Approximate knowledge cutoff

/**
 * Signal 1: Domain Recency.
 * How likely is it that the API has changed since training data cutoff?
 *
 * Based on FreshLLMs (Vu 2023): "fast-changing" vs "slow-changing" facts
 * and our halfLifeDays config from volatile-domains.json.
 */
export function computeDomainRecency(domainName: string): number {
  const domains = loadVolatileDomains();
  const domain = domains.find(d => d.name === domainName);
  if (!domain) return 0.9; // Unknown domain = probably stable

  const daysSinceCutoff = (Date.now() - TRAINING_CUTOFF.getTime()) / (1000 * 60 * 60 * 24);
  const halfLife = domain.halfLifeDays;

  // Exponential decay: confidence halves every halfLifeDays
  // confidence = 0.5 ^ (daysSinceCutoff / halfLifeDays)
  const confidence = Math.pow(0.5, daysSinceCutoff / halfLife);

  // Clamp to [0.05, 0.95] — never fully certain or fully unknown
  return Math.max(0.05, Math.min(0.95, confidence));
}

/**
 * Signal 2: Task Specificity.
 * Generic patterns (React component, fetch call) = high confidence.
 * Specific new APIs (useFormState, streaming responses) = low confidence.
 *
 * Based on Adaptive-RAG (Jeong 2024): route by question complexity.
 */
export function computeSpecificity(taskText: string): number {
  // Indicators of specific API usage (low confidence)
  const specificIndicators = [
    /\bv\d+(\.\d+)+\b/i,           // Version numbers (v4.2.1)
    /\bnew\s+api\b/i,              // Explicit "new API"
    /\b(latest|current|recent)\s+(version|release|update)\b/i,
    /\b(breaking|changed|deprecated|removed)\b/i,
    /\bspecific\s+(endpoint|method|function|parameter)\b/i,
    /\bstraming\b|\bstream\b.*\b(response|completion|message)\b/i,
    /\b(migration|upgrade)\s+(from|to)\b/i,
    /\b(replacement|successor|alternative)\s+(for|to)\b/i,
  ];

  // Indicators of generic patterns (high confidence)
  const genericIndicators = [
    /\b(component|hook|function|class|module)\b/i,
    /\b(basic|simple|standard)\s/i,
    /\b(pattern|approach|architecture|structure)\b/i,
    /\b(refactor|clean|organize)\b/i,
    /\b(CRUD|REST|GraphQL)\b/i,
  ];

  const specificCount = specificIndicators.filter(rx => rx.test(taskText)).length;
  const genericCount = genericIndicators.filter(rx => rx.test(taskText)).length;

  // Score: more specific = lower confidence
  if (specificCount === 0 && genericCount > 0) return 0.9;
  if (specificCount > 0 && genericCount === 0) return 0.3;
  if (specificCount > genericCount) return 0.4;
  if (genericCount > specificCount) return 0.7;
  return 0.6; // Balanced
}

/**
 * Signal 3: Cache Freshness.
 * Do we have verified recent data in the freshness cache?
 *
 * Leverages existing Tier 1 cache from h65-knowledge-freshness-gate.ts.
 */
export function computeCacheFreshness(domainName: string): number {
  try {
    const { isFresh, getLastVerified } = require('../lib/freshness-cache.ts');
    if (isFresh(domainName)) return 1.0;

    // Partially fresh: verified but aging
    const lastVerified = getLastVerified(domainName);
    if (lastVerified) {
      const ageMs = Date.now() - lastVerified;
      const freshnessWindowMs = 7 * 24 * 60 * 60 * 1000;
      // Linear decay from 1.0 (just verified) to 0.3 (2x stale)
      const score = 1.0 - (0.7 * Math.min(1, ageMs / (freshnessWindowMs * 2)));
      return Math.max(0.3, score);
    }

    return 0.2; // No cache entry at all
  } catch {
    return 0.5; // Cache module unavailable — neutral
  }
}

/**
 * Signal 4: Historical Success.
 * Thompson Sampling of past outcomes: did workers succeed or fail on this domain?
 *
 * Based on existing cortex/freshness-thompson.ts infrastructure.
 * Success = worker completed without stale API errors.
 * Failure = worker hit deprecated API, wrong method signature, etc.
 */
export function computeHistoricalSuccess(domainName: string): number {
  try {
    const thompsonPath = heliosPath('extensions', 'cortex', 'data', 'freshness-thompson.json');
    if (!existsSync(thompsonPath)) return 0.5; // No data yet — neutral prior

    const data = JSON.parse(readFileSync(thompsonPath, 'utf8'));
    const entry = data[domainName];
    if (!entry) return 0.5; // No data for this domain

    // Beta distribution: mean = alpha / (alpha + beta)
    const alpha = entry.successes || 1;
    const beta = entry.failures || 1;
    return alpha / (alpha + beta);
  } catch {
    return 0.5; // Neutral prior
  }
}

/**
 * Signal 5: Task Complexity.
 * Simple API call = high confidence (even if API changed, basic usage is similar).
 * Complex multi-step integration = low confidence (more places to go wrong).
 */
export function computeComplexity(taskText: string): number {
  const complexityIndicators = [
    /\b(integrate|integration)\b/i,
    /\b(stream|streaming|websocket|sse)\b/i,
    /\b(auth|oauth|jwt|session)\b/i,
    /\b(middleware|interceptor|plugin)\b/i,
    /\b(multi.?step|pipeline|chain|workflow)\b/i,
    /\b(concurrent|parallel|async|race)\b/i,
    /\b(error.?handl|retry|circuit.?break)\b/i,
  ];

  const matches = complexityIndicators.filter(rx => rx.test(taskText)).length;

  // 0 matches = simple (0.9), 1-2 = moderate (0.6), 3+ = complex (0.3)
  if (matches === 0) return 0.9;
  if (matches <= 2) return 0.6;
  return 0.3;
}

// ── Pre-Generation Assessment ─────────────────────────────────────────

export type PreGenDecision = 'retrieve-first' | 'generate-first' | 'skip';

export interface PreGenResult {
  decision: PreGenDecision;
  confidence: number;
  signals: ConfidenceSignals;
  domains: string[];
  reason: string;
}

/**
 * Phase 1: Pre-generation confidence assessment.
 * Determines whether to retrieve docs before generating, generate then verify, or skip.
 *
 * Key difference from old H65: This NEVER blocks. It only recommends.
 * The decision is advisory — the orchestrator/worker can override.
 */
export function assessPreGeneration(
  taskText: string,
  config: Partial<H65v2Config> = {},
): PreGenResult {
  const cfg = { ...DEFAULT_CONFIG, ...config };

  if (!cfg.preCheck) {
    return { decision: 'skip', confidence: 1.0, signals: neutralSignals(), domains: [], reason: 'preCheck disabled' };
  }

  // Detect domains
  const detection = detectVolatileDomains(taskText);
  if (detection.riskLevel === 'EXEMPT' || detection.riskLevel === 'LOW' || detection.domains.length === 0) {
    return { decision: 'skip', confidence: 1.0, signals: neutralSignals(), domains: [], reason: 'No volatile domains detected' };
  }

  // Compute signals for FIRST detected domain (primary concern)
  const primaryDomain = detection.domains[0];
  const signals: ConfidenceSignals = {
    domainRecency: computeDomainRecency(primaryDomain),
    specificity: computeSpecificity(taskText),
    cacheFreshness: computeCacheFreshness(primaryDomain),
    historicalSuccess: computeHistoricalSuccess(primaryDomain),
    complexity: computeComplexity(taskText),
  };

  const confidence = computeConfidence(signals);

  // Decision routing based on confidence thresholds
  let decision: PreGenDecision;
  let reason: string;

  if (confidence < cfg.retrieveFirstThreshold) {
    decision = 'retrieve-first';
    reason = `Low confidence (${(confidence * 100).toFixed(0)}%) for [${detection.domains.join(', ')}]. ` +
      `Recommend web_search before generation to avoid stale API usage.`;
  } else if (confidence > cfg.skipVerifyThreshold) {
    decision = 'skip';
    reason = `High confidence (${(confidence * 100).toFixed(0)}%) for [${detection.domains.join(', ')}]. ` +
      `Recent verification or stable API — proceed without retrieval.`;
  } else {
    decision = 'generate-first';
    reason = `Moderate confidence (${(confidence * 100).toFixed(0)}%) for [${detection.domains.join(', ')}]. ` +
      `Generate code first, then verify output with post-hoc checks.`;
  }

  return { decision, confidence, signals, domains: detection.domains, reason };
}

// ── Post-Generation Verification ──────────────────────────────────────

export interface VerificationResult {
  passed: boolean;
  checks: VerificationCheck[];
  failedDomains: string[];
  repairHints: string[];
}

export interface VerificationCheck {
  name: string;
  passed: boolean;
  detail: string;
  domain?: string;
}

/**
 * Phase 2: Post-generation verification.
 * Runs AFTER worker generates code. Checks:
 * 1. Import validation — do referenced packages exist in package.json?
 * 2. TypeScript type-check — does `tsc --noEmit` pass?
 * 3. API pattern validation — are known-deprecated patterns used?
 *
 * Key insight from "LLMs Cannot Self-Correct" (Huang 2023):
 * "LLMs struggle to self-correct without external feedback"
 * → We provide EXTERNAL signals: type checker, package manager, static analysis.
 *
 * From "Reflexion" (Shinn 2023):
 * "Language agents with verbal reinforcement learning"
 * → Feed verification failures back as structured repair prompts.
 */
export async function verifyGeneration(
  filePath: string,
  taskText: string,
  config: Partial<H65v2Config> = {},
): Promise<VerificationResult> {
  const cfg = { ...DEFAULT_CONFIG, ...config };
  const checks: VerificationCheck[] = [];
  const failedDomains: string[] = [];
  const repairHints: string[] = [];

  if (!cfg.postVerify || !existsSync(filePath)) {
    return { passed: true, checks: [], failedDomains: [], repairHints: [] };
  }

  const content = readFileSync(filePath, 'utf8');
  const ext = extname(filePath);

  // ── Check 1: Import Validation ────────────────────────────────
  if (cfg.importVerify && ['.ts', '.tsx', '.js', '.jsx', '.mjs'].includes(ext)) {
    const importCheck = await verifyImports(filePath, content);
    checks.push(importCheck);
    if (!importCheck.passed && importCheck.domain) {
      failedDomains.push(importCheck.domain);
      repairHints.push(`Import "${importCheck.detail}" not found in package.json or node_modules. ` +
        `Verify the correct package name and import path for the current version.`);
    }
  }

  // ── Check 2: TypeScript Type Check ────────────────────────────
  if (cfg.typeCheckVerify && ['.ts', '.tsx'].includes(ext)) {
    const typeCheck = await verifyTypes(filePath);
    checks.push(typeCheck);
    if (!typeCheck.passed) {
      repairHints.push(`TypeScript errors found: ${typeCheck.detail}. ` +
        `This may indicate using a deprecated/renamed API. Check current type definitions.`);
    }
  }

  // ── Check 3: Deprecated Pattern Detection ─────────────────────
  const deprecationCheck = checkDeprecatedPatterns(content, taskText);
  checks.push(deprecationCheck);
  if (!deprecationCheck.passed && deprecationCheck.domain) {
    failedDomains.push(deprecationCheck.domain);
    repairHints.push(deprecationCheck.detail);
  }

  const passed = checks.every(c => c.passed);
  return { passed, checks, failedDomains: [...new Set(failedDomains)], repairHints };
}

// ── Verification Helpers ──────────────────────────────────────────────

async function verifyImports(filePath: string, content: string): Promise<VerificationCheck> {
  // Extract import statements
  const importRegex = /(?:import|require)\s*(?:\(?\s*['"]([^'"]+)['"]\s*\)?|.*from\s*['"]([^'"]+)['"])/g;
  const imports: string[] = [];
  let match;
  while ((match = importRegex.exec(content)) !== null) {
    const pkg = match[1] || match[2];
    if (pkg && !pkg.startsWith('.') && !pkg.startsWith('/') && !pkg.startsWith('node:')) {
      // Extract package name (handle scoped packages)
      const pkgName = pkg.startsWith('@') ? pkg.split('/').slice(0, 2).join('/') : pkg.split('/')[0];
      imports.push(pkgName);
    }
  }

  if (imports.length === 0) {
    return { name: 'import-validation', passed: true, detail: 'No external imports' };
  }

  // Check against package.json
  const dir = join(filePath, '..');
  let pkgJson: any = null;
  let searchDir = dir;
  for (let i = 0; i < 5; i++) {
    const pkgPath = join(searchDir, 'package.json');
    if (existsSync(pkgPath)) {
      try { pkgJson = JSON.parse(readFileSync(pkgPath, 'utf8')); } catch {}
      break;
    }
    searchDir = join(searchDir, '..');
  }

  if (!pkgJson) {
    return { name: 'import-validation', passed: true, detail: 'No package.json found — skipping' };
  }

  const allDeps = {
    ...pkgJson.dependencies,
    ...pkgJson.devDependencies,
    ...pkgJson.peerDependencies,
  };

  const missing = imports.filter(imp => !allDeps[imp]);
  if (missing.length === 0) {
    return { name: 'import-validation', passed: true, detail: `All ${imports.length} imports found in package.json` };
  }

  // Check which volatile domain the missing import belongs to
  const domains = loadVolatileDomains();
  const relatedDomain = domains.find(d =>
    missing.some(m => d.name.toLowerCase().includes(m.toLowerCase()) ||
      m.toLowerCase().includes(d.name.toLowerCase()))
  );

  return {
    name: 'import-validation',
    passed: false,
    detail: `Missing packages: [${missing.join(', ')}] — not in package.json`,
    domain: relatedDomain?.name,
  };
}

async function verifyTypes(filePath: string): Promise<VerificationCheck> {
  try {
    const { execSync } = require('node:child_process');
    const result = execSync(`npx tsc --noEmit --pretty false "${filePath}" 2>&1`, {
      timeout: 10000,
      encoding: 'utf8',
      stdio: ['pipe', 'pipe', 'pipe'],
    });
    return { name: 'type-check', passed: true, detail: 'tsc --noEmit passed' };
  } catch (e: any) {
    const output = e.stdout || e.message || '';
    // Extract first 3 errors
    const errors = output.split('\n').filter((l: string) => l.includes('error TS')).slice(0, 3);
    return {
      name: 'type-check',
      passed: false,
      detail: errors.length > 0 ? errors.join('; ') : 'TypeScript compilation failed',
    };
  }
}

/**
 * Check for known deprecated patterns in volatile domains.
 * This is a lightweight static check — no web access needed.
 *
 * Pattern list loaded from config/deprecated-patterns.json (externalized Phase 2).
 * Falls back to built-in patterns if config file is unavailable.
 */
let _deprecatedPatternsCache: Array<{ pattern: RegExp; domain: string; message: string }> | null = null;
let _deprecatedPatternsCacheAt = 0;
const DEPRECATED_PATTERNS_TTL = 60_000; // reload every 60s

function loadDeprecatedPatterns(): Array<{ pattern: RegExp; domain: string; message: string }> {
  if (_deprecatedPatternsCache && (Date.now() - _deprecatedPatternsCacheAt) < DEPRECATED_PATTERNS_TTL) {
    return _deprecatedPatternsCache;
  }
  try {
    const configPath = heliosPath('extensions', 'helios-governance', 'config', 'deprecated-patterns.json');
    if (existsSync(configPath)) {
      const raw = JSON.parse(readFileSync(configPath, 'utf8'));
      const patterns: Array<{ pattern: RegExp; domain: string; message: string }> = [];
      for (const [domain, config] of Object.entries(raw) as [string, { patterns: string[]; docs: string }][]) {
        for (const p of config.patterns) {
          try {
            patterns.push({
              pattern: new RegExp(p, 'i'),
              domain,
              message: `Deprecated pattern detected (${domain}) — see ${config.docs}`,
            });
          } catch { /* skip invalid regex */ }
        }
      }
      _deprecatedPatternsCache = patterns;
      _deprecatedPatternsCacheAt = Date.now();
      return patterns;
    }
  } catch { /* fall through to builtins */ }

  // Fallback: minimal built-in set
  _deprecatedPatternsCache = [
    { pattern: /openai\.ChatCompletion\.create\b/i, domain: 'openai', message: 'openai.ChatCompletion.create is deprecated — use client.chat.completions.create()' },
    { pattern: /\bopenai\.api_key\b/i, domain: 'openai', message: 'Module-level api_key deprecated — use OpenAI(api_key=...) client' },
    { pattern: /\bgetServerSideProps\b/i, domain: 'nextjs', message: 'getServerSideProps is Pages Router — use server components in App Router' },
    { pattern: /\banthropic\.completions?\b/i, domain: 'anthropic', message: 'Legacy completions API — use messages API' },
  ];
  _deprecatedPatternsCacheAt = Date.now();
  return _deprecatedPatternsCache;
}

function checkDeprecatedPatterns(content: string, taskText: string): VerificationCheck {
  const deprecatedPatterns = loadDeprecatedPatterns();

  for (const dep of deprecatedPatterns) {
    if (dep.pattern.test(content)) {
      return {
        name: 'deprecated-pattern',
        passed: false,
        detail: dep.message,
        domain: dep.domain,
      };
    }
  }

  return { name: 'deprecated-pattern', passed: true, detail: 'No known deprecated patterns found' };
}

// ── Thompson Sampling Update ──────────────────────────────────────────

/**
 * Record outcome for Thompson Sampling learning.
 * Called after verification passes or fails to update the prior for each domain.
 *
 * Based on existing cortex/freshness-thompson.ts infrastructure.
 */
export function recordOutcome(domainName: string, success: boolean): void {
  try {
    const thompsonPath = heliosPath('extensions', 'cortex', 'data', 'freshness-thompson.json');
    let data: Record<string, { successes: number; failures: number; lastUpdated: string }> = {};

    if (existsSync(thompsonPath)) {
      data = JSON.parse(readFileSync(thompsonPath, 'utf8'));
    }

    if (!data[domainName]) {
      data[domainName] = { successes: 1, failures: 1, lastUpdated: new Date().toISOString() };
    }

    if (success) {
      data[domainName].successes += 1;
    } else {
      data[domainName].failures += 1;
    }
    data[domainName].lastUpdated = new Date().toISOString();

    // Apply decay (prevent ancient data from dominating)
    const decay = DEFAULT_CONFIG.thompsonDecay;
    data[domainName].successes *= decay;
    data[domainName].failures *= decay;

    // Ensure minimums
    data[domainName].successes = Math.max(1, data[domainName].successes);
    data[domainName].failures = Math.max(1, data[domainName].failures);

    const { writeFileSync, mkdirSync } = require('node:fs');
    mkdirSync(heliosPath('extensions', 'cortex', 'data'), { recursive: true });
    writeFileSync(thompsonPath, JSON.stringify(data, null, 2));
  } catch (e) {
    process.stderr.write(`[h65v2] Thompson update failed (non-fatal): ${String(e)}\n`);
  }
}

// ── Repair Phase ──────────────────────────────────────────────────────

export interface RepairContext {
  /** Original task text */
  taskText: string;
  /** File that failed verification */
  filePath: string;
  /** Verification result with specific failures */
  verification: VerificationResult;
  /** Fresh docs/context fetched after failure */
  freshContext: string;
  /** Repair attempt number (1-indexed) */
  attempt: number;
}

/**
 * Phase 3: Generate repair prompt for the worker.
 * Structures the verification failure into an actionable repair instruction.
 *
 * From Reflexion (Shinn 2023): "verbal reinforcement" is more effective than
 * raw error messages. We structure the feedback as:
 * 1. What failed (specific check)
 * 2. Why it likely failed (stale knowledge hypothesis)
 * 3. Fresh context (fetched docs)
 * 4. Explicit repair instruction
 */
export function generateRepairPrompt(ctx: RepairContext): string {
  const sections: string[] = [];

  sections.push(`## ⚠️ H65v2 Post-Verification Failure — Repair Required (attempt ${ctx.attempt}/${DEFAULT_CONFIG.maxRepairAttempts})`);
  sections.push('');
  sections.push('### What Failed');
  for (const check of ctx.verification.checks.filter(c => !c.passed)) {
    sections.push(`- **${check.name}**: ${check.detail}`);
  }

  sections.push('');
  sections.push('### Likely Cause');
  sections.push('The generated code likely uses stale API patterns from training data.');
  sections.push(`Domains affected: [${ctx.verification.failedDomains.join(', ')}]`);

  if (ctx.freshContext) {
    sections.push('');
    sections.push('### Fresh Context (verified current docs)');
    sections.push(ctx.freshContext);
  }

  sections.push('');
  sections.push('### Repair Instructions');
  for (const hint of ctx.verification.repairHints) {
    sections.push(`- ${hint}`);
  }
  sections.push('');
  sections.push('**Update the code in `' + ctx.filePath + '` to use the CURRENT API as documented above.**');

  return sections.join('\n');
}

// ── Utilities ─────────────────────────────────────────────────────────

function neutralSignals(): ConfidenceSignals {
  return {
    domainRecency: 1.0,
    specificity: 1.0,
    cacheFreshness: 1.0,
    historicalSuccess: 1.0,
    complexity: 1.0,
  };
}

// ── Exports for Integration ───────────────────────────────────────────

export { DEFAULT_CONFIG as H65V2_DEFAULT_CONFIG };
