/**
 * Live-LLM smoke test for the public-chat bot. Fires a small set of canned
 * prompts at the real model with the production system prompt and reports
 * which bot-action types showed up in the responses.
 *
 *   ANTHROPIC_API_KEY=... npx tsx scripts/smoke-public-chat-live.ts
 *
 * The script is OPTIONAL — it costs real API tokens and is non-deterministic.
 * Run it before deploying significant prompt or corpus changes. The vitest
 * suite (`src/public-chat/__tests__/`) covers everything that doesn't need a
 * live model.
 *
 * The script does NOT exit non-zero on a missing action — the LLM is allowed
 * to be cautious. It exits non-zero only on transport errors. Treat the report
 * as a coverage signal: if all five action types reliably surface, the protocol
 * is being honoured. If one or more are repeatedly silent across runs, the
 * persona is probably under-specifying it or the model is dropping it.
 */

import Anthropic from '@anthropic-ai/sdk';
import { buildPublicChatSystem } from '../src/public-chat/system-prompt';

// Minimal extractor — mirrors the frontend's bot-action-parser regex pair
// (BOT_ACTION_REGEX and INVOKE_FORM_REGEX from
// hailer-frontend-2/.../bot-action-parser.ts). Inlined to keep this script
// repo-local with no extra deps.
const ACTION_TYPES = '(navigate|highlight|point|presentation|slide)';
const TAG_REGEXES = [
    new RegExp(`<bot-action\\s+type="${ACTION_TYPES}"\\s+target="([^"]+)"`, 'g'),
    new RegExp(`<invoke\\s+name="bot_action"\\s+type="${ACTION_TYPES}"\\s+target="([^"]+)"`, 'g'),
];

function extractActionTypes(text: string): Set<string> {
    const found = new Set<string>();
    for (const re of TAG_REGEXES) {
        re.lastIndex = 0;
        let m: RegExpExecArray | null;
        while ((m = re.exec(text)) !== null) {
            found.add(m[1]!);
        }
    }
    return found;
}

const MODEL = 'claude-haiku-4-5-20251001';
const MAX_TOKENS = 600;

interface Probe {
    label: string;
    prompt: string;
    /** Action types we'd expect to see at least one of in the reply. */
    expect: Array<'navigate' | 'highlight' | 'point' | 'presentation' | 'slide'>;
}

const PROBES: Probe[] = [
    {
        label: 'tour invitation triggers a guided tour',
        prompt: "I'm a developer evaluating Hailer for our team. Give me a quick walk-through of how this would feel.",
        expect: ['navigate', 'point'],
    },
    {
        label: 'explicit "show me X" triggers navigate or highlight',
        prompt: 'Where do I see the calendar?',
        expect: ['navigate', 'highlight'],
    },
    {
        label: 'overview ask triggers the presentation',
        prompt: "What is Hailer? Give me the elevator pitch — I want the big picture.",
        expect: ['presentation'],
    },
    {
        label: 'concept question triggers an inline slide chip',
        prompt: 'Can you explain what a workflow is in Hailer?',
        expect: ['slide'],
    },
];

async function main(): Promise<void> {
    const apiKey = process.env['ANTHROPIC_API_KEY'];
    if (!apiKey) {
        console.error('ANTHROPIC_API_KEY env var is required for the live smoke.');
        process.exit(2);
    }
    const anthropic = new Anthropic({ apiKey });

    const system = buildPublicChatSystem();

    let probesRun = 0;
    let expectedActionsSatisfied = 0;
    let expectedActionsTotal = 0;
    const misses: Array<{ probe: string; missing: string[] }> = [];

    for (const probe of PROBES) {
        probesRun += 1;
        process.stdout.write(`\n• ${probe.label}\n  prompt: ${probe.prompt}\n`);
        let fullText = '';
        try {
            const response = await anthropic.messages.create({
                model: MODEL,
                max_tokens: MAX_TOKENS,
                temperature: 0.3,
                system,
                messages: [{ role: 'user', content: probe.prompt }],
            });
            for (const block of response.content) {
                if (block.type === 'text') fullText += block.text;
            }
        } catch (err) {
            console.error(`  TRANSPORT ERROR: ${err instanceof Error ? err.message : String(err)}`);
            process.exit(1);
        }

        const surfaced = extractActionTypes(fullText);

        const missing: string[] = [];
        for (const exp of probe.expect) {
            expectedActionsTotal += 1;
            if (surfaced.has(exp)) {
                expectedActionsSatisfied += 1;
            } else {
                missing.push(exp);
            }
        }
        console.log(`  surfaced: ${Array.from(surfaced).join(', ') || '(none)'}`);
        console.log(`  expected: ${probe.expect.join(', ')}`);
        if (missing.length > 0) {
            misses.push({ probe: probe.label, missing });
            console.log(`  MISSING: ${missing.join(', ')}`);
        }
    }

    console.log(
        `\n${expectedActionsSatisfied}/${expectedActionsTotal} expected action types surfaced across ${probesRun} probes`
    );
    if (misses.length > 0) {
        console.log('\nMisses worth investigating if they repeat across runs:');
        for (const m of misses) {
            console.log(`  - "${m.probe}" missing: ${m.missing.join(', ')}`);
        }
    }
    // Don't fail the run on missing actions — the LLM may legitimately reply in
    // plain prose. Transport errors above are the only hard failure.
    process.exit(0);
}

main().catch((err) => {
    console.error(err);
    process.exit(1);
});
