/**
 * crawl4ai-client.ts — TypeScript REST client for crawl4ai Docker service.
 *
 * Service:
 *   docker run -d -p 11235:11235 --name crawl4ai --shm-size=1g \
 *     --restart=unless-stopped unclecode/crawl4ai:latest
 *
 * Architecture: crawl4ai complements SearXNG:
 *   SearXNG  → fast metasearch: discovers relevant URLs + short snippets
 *   crawl4ai → deep scraper: extracts full page content as clean Markdown
 *
 * ── Resilience layers (from the docs) ────────────────────────────────────
 *
 * The crawl4ai v0.8.9+ REST API has a NATIVE anti-bot escalation chain:
 *   CrawlerRunConfig.max_retries   — retry rounds when blocking detected
 *   CrawlerRunConfig.magic         — simulate_user + random delays + overlay removal
 *   BrowserConfig.enable_stealth   — playwright-stealth fingerprint evasion
 *
 * We layer on top with per-domain routing:
 *   Tier 1a — crawl4ai standard (fast, resource-blocked, most domains)
 *   Tier 1b — crawl4ai magic    (bot-protected: Medium, Cloudflare, Stack Overflow)
 *   Tier 2  — Jina Reader       (GitHub + sites that block headless browsers entirely)
 *   Tier 3  — html2text/urllib  (pure Python, zero deps, last resort / offline)
 *
 * Domain strategy table routes URLs to the right tier *before* wasting a round-trip.
 * Domains not in the table get standard mode first, then adaptive magic fallback.
 *
 * Docs consulted:
 *   /advanced/anti-bot-and-fallback/   ← key: max_retries, proxy_config escalation chain
 *   /advanced/undetected-browser/      ← enable_stealth + magic mode
 *   /advanced/identity-based-crawling/ ← user_agent_mode=random, locale, timezone
 *   /advanced/session-management/      ← session_id reuse for sequential crawls
 *   /advanced/multi-url-crawling/      ← MemoryAdaptiveDispatcher, RateLimiter
 *   /core/browser-crawler-config/      ← BrowserConfig fields: avoid_ads, extra_args
 *   /core/self-hosting/                ← hooks disabled by default (needs env var)
 */

// ── Service discovery ────────────────────────────────────────────────────

const CRAWL4AI_CANDIDATES = ['http://localhost:11235', 'http://localhost:11236'];
const CRAWL4AI_ENV = process.env.CRAWL4AI_URL;

let _crawl4aiBase: string | null = null;
let _probeAttempted = false;
let _probeLastAttemptMs = 0;
let _probeInFlight: Promise<string | null> | null = null;
const PROBE_CACHE_TTL_MS = 60_000; // re-probe after 60s if previously failed

// ── Types ─────────────────────────────────────────────────────────────────

/** v0.8.9+: markdown field is an object, not a plain string */
export interface MarkdownObject {
  raw_markdown: string;
  markdown_with_citations?: string;
  references_markdown?: string;
  fit_markdown?: string;
  fit_html?: string;
}

export interface CrawlResult {
  url: string;
  markdown: string | MarkdownObject;
  success: boolean;
  error_message?: string;
  status_code?: number;
}

export interface FetchResult {
  content: string;
  /** Which tier delivered the content */
  source: 'crawl4ai' | 'crawl4ai-magic' | 'jina' | 'trafilatura' | 'none';
  url: string;
  error?: string;
}

// ── Domain strategy routing ───────────────────────────────────────────────
//
// From the anti-bot docs:
//   - Detecting blocking uses HTTP 403/429 + HTML markers (Cloudflare, Akamai, etc.)
//   - start with enable_stealth=True + magic=True to reduce initial blocks
//   - combine with wait_until="load" for anti-bot sites (domcontentloaded fires too early)
//
// Strategy values:
//   'standard'  — headless Chromium, resource-blocked, fast (default)
//   'magic'     — magic=True + simulate_user + random UA + wait_until="load"
//   'jina-only' — skip crawl4ai entirely, go straight to Jina Reader
//
// Sites in this table are routed directly without wasting a failed attempt.

type DomainStrategy = 'standard' | 'magic' | 'jina-only';

const _DOMAIN_STRATEGY: Record<string, DomainStrategy> = {
  // ── GitHub: headless browsers get login walls / incomplete HTML
  // Jina renders GitHub natively with its own rendering pipeline
  'github.com':                'jina-only',
  'raw.githubusercontent.com': 'jina-only',
  'gist.github.com':           'jina-only',
  'docs.github.com':           'jina-only',

  // ── Cloudflare-protected docs — magic mode handles JS challenges
  'docs.anthropic.com':        'magic',
  'platform.openai.com':       'magic',
  'developers.cloudflare.com': 'magic',
  'docs.aws.amazon.com':       'magic',
  'learn.microsoft.com':       'magic',

  // ── Medium / Substack — soft paywalls + bot detection
  'medium.com':                'magic',
  'substack.com':              'magic',
  'towardsdatascience.com':    'magic',

  // ── Stack Overflow / Exchange — aggressive rate limiting + bot detection
  'stackoverflow.com':         'magic',
  'stackexchange.com':         'magic',
  'superuser.com':             'magic',
  'serverfault.com':           'magic',
  'askubuntu.com':             'magic',

  // ── LinkedIn — requires auth; skip to Jina which handles redirects gracefully
  'linkedin.com':              'jina-only',
  'www.linkedin.com':          'jina-only',

  // ── Twitter/X — requires auth for most content; Jina handles better
  'twitter.com':               'jina-only',
  'x.com':                     'jina-only',
};

/** Get the crawl strategy for a URL based on its domain. */
function _getDomainStrategy(url: string): DomainStrategy {
  try {
    const hostname = new URL(url).hostname.replace(/^www\./, '');
    for (const [domain, strategy] of Object.entries(_DOMAIN_STRATEGY)) {
      if (hostname === domain || hostname.endsWith(`.${domain}`)) {
        return strategy;
      }
    }
  } catch { /* malformed URL — fall through */ }
  return 'standard';
}

// ── Markdown extraction ──────────────────────────────────────────────────

/** Extract plain string from markdown field regardless of API version. */
function _extractMarkdown(md: string | MarkdownObject | undefined | null): string {
  if (!md) return '';
  if (typeof md === 'string') return md;
  const raw = md.raw_markdown || '';
  const fit = md.fit_markdown || '';
  // Prefer fit_markdown when it's substantial relative to raw (≥60% of raw length).
  // fit_markdown is pruned of nav/sidebar remnants that excluded_tags doesn't fully catch.
  // Fall back to raw when fit over-pruned (< 60% of raw) — e.g. dense code-heavy docs pages.
  // Citations markdown is last resort (usually shorter and reference-heavy).
  if (fit && raw && fit.length >= raw.length * 0.60) return fit;
  return raw || fit || md.markdown_with_citations || '';
}

// ── Content truncation ───────────────────────────────────────────────────

const _FETCH_MAX_CHARS = 40_000; // ~10K tokens @ ~4 chars/token

function _truncateContent(text: string, maxChars = _FETCH_MAX_CHARS): string {
  if (text.length <= maxChars) return text;
  return (
    text.slice(0, maxChars) +
    `\n\n[CONTENT TRUNCATED: ${text.length.toLocaleString()} chars total — kept first ${maxChars.toLocaleString()}]`
  );
}

// ── Service discovery ────────────────────────────────────────────────────

async function _discoverCrawl4AI(): Promise<string | null> {
  if (_crawl4aiBase) return _crawl4aiBase;
  if (CRAWL4AI_ENV) { _crawl4aiBase = CRAWL4AI_ENV; return _crawl4aiBase; }
  if (_probeAttempted && Date.now() - _probeLastAttemptMs < PROBE_CACHE_TTL_MS) return null;

  if (_probeInFlight) return _probeInFlight;
  _probeInFlight = _doCrawl4AIProbe().finally(() => { _probeInFlight = null; });
  return _probeInFlight;
}

async function _doCrawl4AIProbe(): Promise<string | null> {
  _probeAttempted = true;
  _probeLastAttemptMs = Date.now();
  for (const base of CRAWL4AI_CANDIDATES) {
    try {
      const r = await fetch(`${base}/health`, { signal: AbortSignal.timeout(3000) });
      if (r.ok) {
        _crawl4aiBase = base;
        process.stderr.write(`[crawl4ai] discovered at ${base}\n`);
        return base;
      }
    } catch { /* try next */ }
  }
  return null;
}

// ── Crawl4AI request builders ─────────────────────────────────────────────
//
// Key insights from the docs:
//
// BrowserConfig fields (passed as browser_config in REST body):
//   headless          — True for server use
//   browser_type      — "chromium" (default), "firefox", "webkit"
//   enable_stealth    — playwright-stealth fingerprint patches (beats navigator.webdriver etc.)
//   user_agent_mode   — "random" rotates UA per request
//   extra_args        — ["--disable-blink-features=AutomationControlled"] removes automation flag
//   avoid_ads         — blocks Google Analytics, DoubleClick, Facebook Pixel at context level
//
// CrawlerRunConfig fields (passed as crawler_config in REST body):
//   cache_mode        — "bypass" for fresh content
//   word_count_threshold — filter out tiny blocks
//   excluded_tags     — strip nav/footer/scripts before markdown generation
//   magic             — True: simulate_user + random timings + overlay removal (from identity docs)
//   simulate_user     — realistic mouse movements and human-like delays
//   remove_overlay_elements — dismiss cookie banners and popups
//   wait_until        — "load" waits for full page load (anti-bot docs: use this not domcontentloaded)
//   delay_before_return_html — extra settle time for JS-rendered pages
//   page_timeout      — generous timeout for slow anti-bot challenge pages
//   max_retries       — rounds of retry when blocking detected (anti-bot native escalation)
//
// NOTE: hooks are DISABLED by default in the Docker image.
// Sending a hooks field returns HTTP 403. Use excluded_tags + avoid_ads instead.
// Use c4a_script (C4A-Script DSL) for structured page interaction — it compiles
// to Playwright JS internally and works without any env var configuration.

// ── C4A-Script programs ───────────────────────────────────────────────────
//
// C4A-Script is a declarative browser automation DSL supported by crawl4ai v0.8+.
// It runs BEFORE HTML is captured, so it can expand lazy-loaded content,
// dismiss overlays, and scroll infinite feeds — all before extraction.
//
// Key commands used:
//   IF (EXISTS `selector`) THEN command   — handle optional elements safely
//   REPEAT (SCROLL DOWN n, count)         — trigger lazy loading
//   WAIT `selector` timeout               — wait for content to appear
//   EVAL `js`                             — run arbitrary JavaScript
//   PROC name ... ENDPROC                 — reusable command blocks
//
// Docs: https://docs.crawl4ai.com/api/c4a-script-reference/

/**
 * Standard C4A-Script — runs on every standard crawl.
 * Goals:
 *   1. Dismiss cookie/GDPR/newsletter overlays BEFORE extraction
 *      (they pollute markdown with consent text instead of page content)
 *   2. Scroll the viewport twice to trigger lazy-loaded images and text
 *   3. Handle "Load More" buttons for paginated content
 *
 * Uses IF (EXISTS ...) so it's safe on pages that don't have these elements.
 */
const _STANDARD_C4A_SCRIPT = `
# ── Cookie / GDPR banner dismissal ──────────────────────────────────────────
# Each IF (EXISTS) is safe to run even if the element is absent.
# Covers the most common cookie banner patterns across sites.
IF (EXISTS \`[id*="cookie"][id*="accept"], [id*="cookie"][id*="agree"]\`) THEN CLICK \`[id*="cookie"][id*="accept"], [id*="cookie"][id*="agree"]\`
IF (EXISTS \`[class*="cookie"][class*="accept"], [class*="cookie"][class*="agree"]\`) THEN CLICK \`[class*="cookie"][class*="accept"], [class*="cookie"][class*="agree"]\`
IF (EXISTS \`button[aria-label*="Accept"], button[aria-label*="Agree"]\`) THEN CLICK \`button[aria-label*="Accept"], button[aria-label*="Agree"]\`
IF (EXISTS \`#onetrust-accept-btn-handler\`) THEN CLICK \`#onetrust-accept-btn-handler\`
IF (EXISTS \`.cc-accept, .cc-dismiss, .cc-btn-accept\`) THEN CLICK \`.cc-accept, .cc-dismiss, .cc-btn-accept\`
IF (EXISTS \`[data-testid*="accept-cookie"], [data-testid*="cookie-accept"]\`) THEN CLICK \`[data-testid*="accept-cookie"], [data-testid*="cookie-accept"]\`

# ── Newsletter / subscription modal dismissal ────────────────────────────────
IF (EXISTS \`[class*="newsletter"][class*="close"], [class*="modal"][class*="close"]\`) THEN CLICK \`[class*="newsletter"][class*="close"], [class*="modal"][class*="close"]\`
IF (EXISTS \`[aria-label="Close"], [aria-label="close modal"]\`) THEN CLICK \`[aria-label="Close"], [aria-label="close modal"]\`

# ── Scroll to trigger lazy loading ───────────────────────────────────────────
# Scroll down in two steps then back up so the full page is visible.
# This triggers IntersectionObserver-based lazy loaders.
SCROLL DOWN 600
WAIT 1
SCROLL DOWN 1200
WAIT 1
SCROLL UP 1800
`.trim();

/**
 * Magic C4A-Script — deeper interaction for bot-protected and JS-heavy sites.
 * Adds on top of the standard script:
 *   1. Scroll the full page height to load all lazy content
 *   2. Click "Load More" / "Show More" / pagination buttons if present
 *   3. Wait for dynamic content after each interaction
 *   4. Handle sticky interstitials (age gates, paywalls prompts, etc.)
 */
const _MAGIC_C4A_SCRIPT = `
# ── Interstitial / age gate / overlay dismissal ──────────────────────────────
IF (EXISTS \`[id*="cookie"][id*="accept"], [id*="cookie"][id*="agree"]\`) THEN CLICK \`[id*="cookie"][id*="accept"], [id*="cookie"][id*="agree"]\`
IF (EXISTS \`[class*="cookie"][class*="accept"]\`) THEN CLICK \`[class*="cookie"][class*="accept"]\`
IF (EXISTS \`#onetrust-accept-btn-handler\`) THEN CLICK \`#onetrust-accept-btn-handler\`
IF (EXISTS \`button[aria-label*="Accept"]\`) THEN CLICK \`button[aria-label*="Accept"]\`
IF (EXISTS \`.cc-accept, .cc-dismiss\`) THEN CLICK \`.cc-accept, .cc-dismiss\`
IF (EXISTS \`[class*="modal"][class*="close"]\`) THEN CLICK \`[class*="modal"][class*="close"]\`
IF (EXISTS \`[aria-label="Close"]\`) THEN CLICK \`[aria-label="Close"]\`

# ── Scroll full page to trigger all lazy-loaded content ──────────────────────
REPEAT (SCROLL DOWN 800, 4)
WAIT 2

# ── Load More / Show More / infinite scroll trigger ──────────────────────────
# Click load-more buttons up to 3 times to get deeper content.
IF (EXISTS \`[class*="load-more"], [id*="load-more"]\`) THEN CLICK \`[class*="load-more"], [id*="load-more"]\`
WAIT 2
IF (EXISTS \`[class*="load-more"], [id*="load-more"]\`) THEN CLICK \`[class*="load-more"], [id*="load-more"]\`
WAIT 2
IF (EXISTS \`[class*="show-more"], button[class*="more"]\`) THEN CLICK \`[class*="show-more"], button[class*="more"]\`
WAIT 1

# ── Final scroll to bottom to capture everything ─────────────────────────────
EVAL \`window.scrollTo(0, document.body.scrollHeight)\`
WAIT 1
SCROLL UP 2000
`.trim();

// ── Proxy config (MED-6) ──────────────────────────────────────────────────
/**
 * Build a proxy_config object from the CRAWL4AI_PROXY env var.
 * Supports "http://user:pass@host:port" or plain "host:port" format.
 * Returns undefined when the env var is not set (no proxy — default behaviour).
 *
 * Set CRAWL4AI_PROXY when IP-level blocking occurs (Cloudflare, Akamai) and
 * magic mode + Jina both fail. A residential proxy bypasses datacenter IP blocks.
 *
 * @example
 *   export CRAWL4AI_PROXY="http://user:pass@residential-proxy.example.com:9090"
 */
function _proxyConfig(): { server: string; username?: string; password?: string } | undefined {
  const raw = (process.env.CRAWL4AI_PROXY ?? '').trim();
  if (!raw) return undefined;
  try {
    const u = new URL(raw);
    return {
      server: `${u.protocol}//${u.hostname}:${u.port}`,
      ...(u.username ? { username: decodeURIComponent(u.username) } : {}),
      ...(u.password ? { password: decodeURIComponent(u.password) } : {}),
    };
  } catch {
    // Plain "host:port" without protocol
    return { server: raw.startsWith('http') ? raw : `http://${raw}` };
  }
}

/** Standard crawl config — fast, efficient, suitable for most pages. */
function _standardConfig() {
  return {
    browser_config: {
      headless: true,
      browser_type: 'chromium',
      user_agent: 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36',
      extra_args: ['--disable-blink-features=AutomationControlled', '--no-sandbox', '--disable-dev-shm-usage'],
      viewport_width: 1280,
      viewport_height: 800,
    },
    crawler_config: {
      cache_mode: 'bypass',
      word_count_threshold: 10,
      excluded_tags: ['script', 'style', 'noscript', 'iframe', 'nav', 'footer', 'aside'],
      wait_until: 'domcontentloaded',
      page_timeout: 30000,
      // scan_full_page: scroll entire page to trigger lazy-loaded content before extraction.
      // Works alongside c4a_script scrolling — belt-and-suspenders for IntersectionObserver loaders.
      scan_full_page: true,
      // scroll_delay: pause between scroll steps so IntersectionObserver loaders have time to fire.
      // Without this, scan_full_page scrolls too fast for lazy-loaded text sections to appear.
      scroll_delay: 0.3,
      // wait_for_images: resolve all img src/data-src before capture.
      // Prevents "data-src=blob:..." garbage appearing in extracted markdown.
      wait_for_images: true,
      // remove_consent_popups: native CMP-aware popup removal (OneTrust, Cookiebot, Didomi).
      // More targeted than C4A-Script IF(EXISTS) selectors — handles registered CMP patterns.
      remove_consent_popups: true,
      // Strip outbound links from markdown — they're noise for RESEARCH phase content.
      // Reduces token count and improves signal density on blog/docs pages.
      exclude_external_links: true,
      exclude_social_media_links: true,
      // Proxy routing when CRAWL4AI_PROXY env var is set (e.g. residential proxy to bypass IP blocks)
      ...((_proxyConfig()) ? { proxy_config: _proxyConfig() } : {}),
      // C4A-Script: declarative browser automation DSL that runs BEFORE HTML capture.
      // Uses structured control flow (IF/EXISTS/REPEAT/PROC) to:
      //   1. Dismiss cookie/GDPR banners so they don't pollute the extracted markdown
      //   2. Scroll the viewport to trigger lazy-loaded images and text
      // NOTE: Safe in the default Docker image — c4a_script compiles to Playwright JS
      // internally, NOT the hooks API (which requires CRAWL4AI_HOOKS_ENABLED env var).
      c4a_script: _STANDARD_C4A_SCRIPT,
    },
  };
}

/**
 * Magic-mode config — uses crawl4ai's native anti-bot system:
 *   magic=True         → simulate_user + random interaction timing + auto overlay dismiss
 *   enable_stealth=True → playwright-stealth patches (navigator.webdriver etc.)
 *   user_agent_mode=random → different UA per request
 *   wait_until="load"  → critical for anti-bot: wait for full page including JS sensors
 *   max_retries=1      → one automatic retry round on 403/CAPTCHA detection
 *
 * From docs: "Combine stealth + magic for maximum evasion" and
 * "wait_until=load is important for anti-bot sites"
 */
function _magicConfig() {
  return {
    browser_config: {
      headless: true,
      browser_type: 'chromium',
      // enable_stealth uses playwright-stealth — beats WebDriver detection, plugin emulation etc.
      enable_stealth: true,
      // random UA rotation prevents fingerprinting across requests
      user_agent_mode: 'random',
      extra_args: ['--disable-blink-features=AutomationControlled', '--no-sandbox', '--disable-dev-shm-usage'],
      viewport_width: 1440,
      viewport_height: 900,
    },
    crawler_config: {
      cache_mode: 'bypass',
      word_count_threshold: 10,
      excluded_tags: ['script', 'style', 'noscript', 'iframe'],
      // magic=True: the single most important flag — enables full simulate_user pipeline
      magic: true,
      // simulate_user: realistic mouse movements + human-like delays (works with magic=True)
      simulate_user: true,
      // remove_overlay_elements: generic overlay detection (belt-and-suspenders with below)
      remove_overlay_elements: true,
      // remove_consent_popups: native CMP-aware removal (OneTrust, Cookiebot, Didomi).
      // More reliable than CSS selector guessing for registered CMPs.
      remove_consent_popups: true,
      // Strip outbound/social links — noise for RESEARCH extraction
      exclude_external_links: true,
      exclude_social_media_links: true,
      // flatten_shadow_dom: resolve Web Component shadow trees before markdown generation.
      // Sites built with Stencil, Lit, Angular Elements render inside shadow roots —
      // without this you get near-empty markdown on enterprise/SaaS docs and product pages.
      flatten_shadow_dom: true,
      // CRITICAL for anti-bot: wait for full page load including JS anti-bot sensors
      wait_until: 'load',
      // Extra settle time for JS-rendered pages after load fires
      delay_before_return_html: 2.0,
      // Generous timeout for anti-bot challenge pages
      page_timeout: 45000,
      // Native retry: crawl4ai detects 403/challenge pages and retries automatically
      // From docs: max_retries=1 gives (1+1)*1 = 2 attempts before giving up
      max_retries: 1,
      // Proxy routing when CRAWL4AI_PROXY env var is set
      ...((_proxyConfig()) ? { proxy_config: _proxyConfig() } : {}),
      // C4A-Script: deeper interaction — scroll full page, click Load More, dismiss overlays.
      // Runs BEFORE HTML capture so lazy content, pagination, and banners are all handled.
      c4a_script: _MAGIC_C4A_SCRIPT,
    },
  };
}

// ── Core POST helper ──────────────────────────────────────────────────────

async function _postCrawl(
  base: string,
  url: string,
  configs: { browser_config: object; crawler_config: object },
  timeoutMs: number,
): Promise<string | null> {
  try {
    const resp = await fetch(`${base}/crawl`, {
      method: 'POST',
      headers: {
        'Content-Type': 'application/json',
        ...(process.env.CRAWL4AI_API_TOKEN
          ? { Authorization: `Bearer ${process.env.CRAWL4AI_API_TOKEN}` }
          : {}),
      },
      body: JSON.stringify({
        urls: [url],
        browser_config: configs.browser_config,
        crawler_config: configs.crawler_config,
      }),
      signal: AbortSignal.timeout(timeoutMs),
    });

    if (!resp.ok) {
      process.stderr.write(`[crawl4ai] HTTP ${resp.status} for ${url.slice(0, 60)}\n`);
      return null;
    }

    const data = await resp.json() as {
      results?: Array<{ success?: boolean; markdown?: string | MarkdownObject; error_message?: string }>;
    };
    const r = data.results?.[0];
    const md = _extractMarkdown(r?.markdown);

    if (!r?.success) {
      process.stderr.write(`[crawl4ai] success=false: ${(r?.error_message ?? '').slice(0, 80)} — ${url.slice(0, 60)}\n`);
      return null;
    }
    // 50-char floor: only reject truly empty results.
    // example.com legitimately returns ~166 chars of clean markdown —
    // the old 200-char threshold caused unnecessary fallback to trafilatura.
    if (md.length < 50) {
      process.stderr.write(`[crawl4ai] markdown empty (${md.length} chars) — ${url.slice(0, 60)}\n`);
      return null;
    }

    return md;
  } catch (e) {
    process.stderr.write(`[crawl4ai] request error: ${String(e).slice(0, 80)} — ${url.slice(0, 60)}\n`);
    return null;
  }
}

// ── Tier implementations ──────────────────────────────────────────────────

async function _fetchViaCrawl4AIStandard(base: string, url: string): Promise<string | null> {
  return _postCrawl(base, url, _standardConfig(), 30_000);
}

/**
 * Magic mode: uses crawl4ai's native anti-bot escalation.
 * Timeout is higher (55s) to account for: page load + JS settle + up to 1 automatic retry.
 * From docs: max_retries=1 means crawl4ai tries twice internally on blocking detection.
 */
async function _fetchViaCrawl4AIMagic(base: string, url: string): Promise<string | null> {
  process.stderr.write(`[crawl4ai] magic mode (stealth+simulate_user+max_retries=1): ${url.slice(0, 60)}\n`);
  return _postCrawl(base, url, _magicConfig(), 55_000);
}

/**
 * Jina Reader — external service that renders pages with its own headless pipeline.
 * Handles GitHub (uses its own rendering), paywalled sites, and heavy SPAs.
 * Free tier: ~200 requests/day without auth.
 * Set JINA_API_KEY env var for higher limits.
 */
async function _fetchViaJina(url: string, maxTokens = 8000): Promise<string | null> {
  try {
    const jinaKey = process.env.JINA_API_KEY;
    const resp = await fetch(`https://r.jina.ai/${url}`, {
      headers: {
        'X-Max-Tokens': String(maxTokens),
        'X-Retain-Images': 'none',
        'X-Retain-Links': 'text',
        ...(jinaKey ? { Authorization: `Bearer ${jinaKey}` } : {}),
      },
      signal: AbortSignal.timeout(20_000),
    });
    if (!resp.ok) return null;
    const text = await resp.text();
    return text.length > 50 ? text : null;
  } catch {
    return null;
  }
}

/**
 * Tier 3: html2text + urllib.request — pure Python, zero binary/C-extension deps.
 * Falls back to stdlib html.parser if html2text (apt: python3-html2text) is unavailable.
 * Works fully offline.
 * Set TRAFILATURA_PYTHON env var to override the Python binary (default: 'python3' on PATH).
 */
async function _fetchViaTrafilatura(url: string): Promise<string | null> {
  const { spawn } = await import('child_process');
  const pythonBin = process.env.TRAFILATURA_PYTHON ?? 'python3';
  const script = [
    'import sys, urllib.request, html.parser',
    'try:',
    '    import html2text as _h2t',
    '    _cv = _h2t.HTML2Text()',
    '    _cv.ignore_links = False',
    '    _cv.ignore_images = True',
    '    _cv.body_width = 0',
    'except ImportError:',
    '    _cv = None',
    'try:',
    '    _ua = "Mozilla/5.0 (compatible; helios-fetch/1.0)"',
    '    _req = urllib.request.Request(sys.argv[1], headers={"User-Agent": _ua})',
    '    with urllib.request.urlopen(_req, timeout=10) as _r:',
    '        _html = _r.read(500_000).decode("utf-8", errors="replace")',
    '    if _cv:',
    '        print(_cv.handle(_html), end="")',
    '    else:',
    '        class _S(html.parser.HTMLParser):',
    '            def __init__(self): super().__init__(); self._t=[]; self._sk=False',
    '            def handle_starttag(self,t,a): self._sk=t in("script","style","noscript")',
    '            def handle_endtag(self,t):',
    '                if t in("script","style","noscript"): self._sk=False',
    '            def handle_data(self,d):',
    '                if not self._sk: self._t.append(d)',
    '        _p=_S(); _p.feed(_html); print(" ".join(_p._t), end="")',
    'except Exception: sys.exit(1)',
  ].join('\n');

  return new Promise((resolve) => {
    const proc = spawn(pythonBin, ['-c', script, url], { timeout: 15_000, env: { ...process.env } });
    let out = '';
    proc.stdout?.on('data', (d: Buffer) => { out += d.toString(); });
    proc.on('close', (code: number | null) => resolve(code === 0 && out.length > 100 ? out : null));
    proc.on('error', () => resolve(null));
  });
}

// ── Public API ─────────────────────────────────────────────────────────────

/**
 * Batch-crawl a list of URLs returning Markdown content.
 * Applies domain strategy routing per-URL.
 * Fail-open: returns empty array if crawl4ai is unavailable.
 */
export async function crawlUrls(
  urls: string[],
  opts: { timeoutMs?: number; maxUrls?: number } = {}
): Promise<Array<{ url: string; markdown: string; success: boolean }>> {
  const { maxUrls = 2 } = opts;
  const limited = urls.slice(0, maxUrls);
  if (limited.length === 0) return [];

  const base = await _discoverCrawl4AI();
  if (!base) return [];

  const results: Array<{ url: string; markdown: string; success: boolean }> = [];

  for (const url of limited) {
    const strategy = _getDomainStrategy(url);

    // jina-only: skip crawl4ai for domains known to block headless browsers
    if (strategy === 'jina-only') {
      const jina = await _fetchViaJina(url, 8000);
      if (jina && jina.length > 50) {
        results.push({ url, markdown: jina, success: true });
        process.stderr.write(`[crawl4ai] jina-only OK: ${url.slice(0, 60)} (${jina.length} chars)\n`);
      }
      continue;
    }

    // magic: go straight to magic mode for known bot-protected domains
    if (strategy === 'magic') {
      const md = await _fetchViaCrawl4AIMagic(base, url);
      if (md && md.length > 50) {
        results.push({ url, markdown: md, success: true });
        continue;
      }
      // magic failed — try Jina as fallback
      const jina = await _fetchViaJina(url, 8000);
      if (jina && jina.length > 50) {
        results.push({ url, markdown: jina, success: true });
      }
      continue;
    }

    // standard: try fast mode first, adaptive magic if it fails
    const md = await _fetchViaCrawl4AIStandard(base, url);
    if (md && md.length > 50) {
      results.push({ url, markdown: md, success: true });
      continue;
    }
    // Adaptive: standard failed — try magic as fallback
    const magicMd = await _fetchViaCrawl4AIMagic(base, url);
    if (magicMd && magicMd.length > 50) {
      results.push({ url, markdown: magicMd, success: true });
    }
  }

  return results;
}

/**
 * Fetch a single URL's content through the full resilience stack.
 *
 * Decision tree based on domain strategy:
 *
 *   jina-only domains (github.com, linkedin.com, twitter.com):
 *     → Jina directly → trafilatura
 *
 *   magic domains (medium.com, stackoverflow.com, Cloudflare docs):
 *     → crawl4ai magic (stealth + max_retries=1) → Jina → trafilatura
 *
 *   standard domains (everything else):
 *     → crawl4ai standard → crawl4ai magic (adaptive) → Jina → trafilatura
 *
 * The crawl4ai magic mode uses the NATIVE anti-bot escalation from the docs:
 *   enable_stealth=True, magic=True, simulate_user=True, wait_until="load", max_retries=1
 * This gives crawl4ai itself 2 internal attempts before we move to the next tier.
 */
export async function fetchUrlContent(
  url: string,
  opts: { maxTokens?: number } = {},
): Promise<FetchResult> {
  const maxTokens = opts.maxTokens ?? 8_000;
  const maxChars = maxTokens * 4;
  const strategy = _getDomainStrategy(url);

  // ── Tier 1: crawl4ai ──────────────────────────────────────────────────
  if (strategy !== 'jina-only') {
    const base = await _discoverCrawl4AI();
    if (base) {
      if (strategy === 'standard') {
        // 1a: fast standard mode
        const r = await _fetchViaCrawl4AIStandard(base, url);
        if (r) {
          process.stderr.write(`[fetch-stack] crawl4ai[standard] OK: ${url.slice(0, 60)} (${r.length} chars)\n`);
          return { content: _truncateContent(r, maxChars), source: 'crawl4ai', url };
        }
        process.stderr.write(`[fetch-stack] crawl4ai[standard] failed, trying magic: ${url.slice(0, 60)}\n`);
      }

      // 1b: magic mode (for magic-strategy domains, or as adaptive fallback after standard fails)
      // Includes native anti-bot escalation: stealth + simulate_user + max_retries=1
      const magicR = await _fetchViaCrawl4AIMagic(base, url);
      if (magicR) {
        process.stderr.write(`[fetch-stack] crawl4ai[magic] OK: ${url.slice(0, 60)} (${magicR.length} chars)\n`);
        return { content: _truncateContent(magicR, maxChars), source: 'crawl4ai-magic', url };
      }
      process.stderr.write(`[fetch-stack] crawl4ai[magic] failed, trying Jina: ${url.slice(0, 60)}\n`);
    } else {
      process.stderr.write(`[fetch-stack] crawl4ai unavailable, trying Jina: ${url.slice(0, 60)}\n`);
    }
  }

  // ── Tier 2: Jina Reader ───────────────────────────────────────────────
  // Handles GitHub, heavy SPAs, and any site that blocks headless browsers.
  const jina = await _fetchViaJina(url, maxTokens);
  if (jina) {
    process.stderr.write(`[fetch-stack] Jina OK: ${url.slice(0, 60)} (${jina.length} chars)\n`);
    return { content: _truncateContent(jina, maxChars), source: 'jina', url };
  }
  process.stderr.write(`[fetch-stack] Jina failed, trying trafilatura: ${url.slice(0, 60)}\n`);

  // ── Tier 3: html2text/urllib ──────────────────────────────────────────
  const traf = await _fetchViaTrafilatura(url);
  if (traf) {
    process.stderr.write(`[fetch-stack] trafilatura OK: ${url.slice(0, 60)} (${traf.length} chars)\n`);
    return { content: _truncateContent(traf, maxChars), source: 'trafilatura', url };
  }

  process.stderr.write(`[fetch-stack] all tiers exhausted for: ${url.slice(0, 60)}\n`);
  return { content: '', source: 'none', url, error: 'All fetch tiers exhausted' };
}

// ── Cache and config management ───────────────────────────────────────────

/** Reset discovery cache — call after container restart or for testing. */
export function resetCrawl4AICache(): void {
  _crawl4aiBase = null;
  _probeAttempted = false;
  _probeInFlight = null;
}

/**
 * Register or override a domain's crawl strategy at runtime.
 * Useful from tests or environment-specific config.
 *
 * @example
 *   // Make notion.so use magic mode
 *   registerDomainStrategy('notion.so', 'magic');
 *   // Reset a domain back to standard
 *   registerDomainStrategy('example.com', 'standard');
 *
 * @param domain   Bare domain, e.g. 'github.com' (www. prefix stripped automatically)
 * @param strategy 'standard' | 'magic' | 'jina-only'
 */
export function registerDomainStrategy(domain: string, strategy: DomainStrategy): void {
  const bare = domain.replace(/^www\./, '');
  if (strategy === 'standard') {
    delete (_DOMAIN_STRATEGY as Record<string, DomainStrategy>)[bare];
  } else {
    (_DOMAIN_STRATEGY as Record<string, DomainStrategy>)[bare] = strategy;
  }
  process.stderr.write(`[crawl4ai] domain strategy: ${bare} → ${strategy}\n`);
}

/**
 * Get the current strategy for a URL (useful for debugging and tests).
 *
 * @example
 *   getDomainStrategy('https://github.com/foo/bar') // → 'jina-only'
 *   getDomainStrategy('https://medium.com/article')  // → 'magic'
 *   getDomainStrategy('https://example.com')          // → 'standard'
 */
export function getDomainStrategy(url: string): DomainStrategy {
  return _getDomainStrategy(url);
}

/**
 * Fetch a URL using a custom C4A-Script for deep page interaction.
 *
 * Use this when you need to:
 *   - Log into a site before extracting content
 *   - Navigate through multi-step flows
 *   - Handle infinite scroll / pagination beyond the built-in magic scroll
 *   - Interact with SPAs that require specific user actions
 *
 * C4A-Script runs BEFORE HTML capture. The script uses a declarative DSL:
 *   IF (EXISTS `selector`) THEN command
 *   REPEAT (SCROLL DOWN 800, 5)
 *   WAIT `selector` timeout
 *   CLICK `selector`
 *   TYPE "text"
 *   EVAL `javascript`
 *   PROC name ... ENDPROC
 *
 * Reference: https://docs.crawl4ai.com/api/c4a-script-reference/
 *
 * @example
 * // Scroll an infinite feed before extracting
 * await crawlWithScript('https://feeds.example.com', `
 *   REPEAT (SCROLL DOWN 800, `document.querySelector(".load-more")`)
 *   WAIT 2
 * `);
 *
 * @example
 * // Login then extract protected page
 * await crawlWithScript('https://app.example.com/dashboard', `
 *   IF (NOT EXISTS \`.user-menu\`) THEN GO /login
 *   WAIT \`#login-form\` 5
 *   CLICK \`#email\`
 *   TYPE "user@example.com"
 *   CLICK \`#password\`
 *   TYPE "password123"
 *   CLICK \`button[type="submit"]\`
 *   WAIT \`.dashboard\` 10
 * `);
 */
export async function crawlWithScript(
  url: string,
  c4aScript: string,
  opts: {
    maxTokens?: number;
    useMagic?: boolean;
    /** JS to execute BEFORE the wait_for check fires — use to click tab/accordion to reveal content */
    jsBeforeWait?: string;
    /** CSS selector or JS expression to wait for before HTML capture (e.g. "#content-loaded") */
    waitFor?: string;
  } = {},
): Promise<FetchResult> {
  const { maxTokens = 8_000, useMagic = false, jsBeforeWait, waitFor } = opts;
  const maxChars = maxTokens * 4;

  const base = await _discoverCrawl4AI();
  if (!base) {
    process.stderr.write(`[crawl4ai] crawlWithScript: service unavailable, falling back to fetchUrlContent\n`);
    return fetchUrlContent(url, { maxTokens });
  }

  const baseConfig = useMagic ? _magicConfig() : _standardConfig();
  const mergedConfig = {
    ...baseConfig,
    crawler_config: {
      ...baseConfig.crawler_config,
      // Override with caller's script — replaces the built-in standard/magic scripts
      c4a_script: c4aScript,
      // Always use load for script-driven crawls (scripts may need full page)
      wait_until: 'load',
      page_timeout: useMagic ? 60000 : 45000,
      // js_code_before_wait: runs JS BEFORE wait_for fires — pipeline order:
      //   navigation → js_code_before_wait → wait_for → delay_before_return_html → capture
      // Use for: click-to-reveal tabs, accordion expansion, SPA navigation triggers
      ...(jsBeforeWait ? { js_code_before_wait: jsBeforeWait } : {}),
      // wait_for: CSS selector or JS expression to await before extraction
      ...(waitFor ? { wait_for: waitFor } : {}),
    },
  };

  process.stderr.write(`[crawl4ai] crawlWithScript (${useMagic ? 'magic' : 'standard'}): ${url.slice(0, 60)}\n`);
  const result = await _postCrawl(base, url, mergedConfig, useMagic ? 65_000 : 50_000);

  if (result) {
    process.stderr.write(`[fetch-stack] crawlWithScript OK: ${url.slice(0, 60)} (${result.length} chars)\n`);
    return { content: _truncateContent(result, maxChars), source: 'crawl4ai', url };
  }

  // Fallback to normal tiered fetch if the script approach fails
  process.stderr.write(`[fetch-stack] crawlWithScript failed, falling back to fetchUrlContent: ${url.slice(0, 60)}\n`);
  return fetchUrlContent(url, { maxTokens });
}

/**
 * Pre-built C4A-Script constants exposed for callers to compose or extend.
 *
 * @example
 * // Add login before the standard banner/scroll script
 * const myScript = C4A_SCRIPTS.LOGIN_STUB + '\n' + C4A_SCRIPTS.STANDARD;
 * await crawlWithScript(url, myScript);
 */
export const C4A_SCRIPTS = {
  /** Standard cookie dismissal + lazy-load scroll — same as the default standard config */
  STANDARD: _STANDARD_C4A_SCRIPT,
  /** Magic deep scroll + load-more + full-page extraction — same as the default magic config */
  MAGIC: _MAGIC_C4A_SCRIPT,
  /** Reusable login procedure stub — fill in your selectors and credentials */
  LOGIN_STUB: `
# Login flow — customise selectors and credentials for your site
SETVAR email = "user@example.com"
SETVAR password = "REPLACE_ME"
IF (NOT EXISTS \`.user-menu\`) THEN GO /login
WAIT \`#login-form\` 8
CLICK \`#email\`
TYPE $email
PRESS Tab
TYPE $password
CLICK \`button[type="submit"]\`
WAIT \`.dashboard, .home, .feed\` 10
`.trim(),
  /** Infinite scroll — scrolls until no more "load more" button exists */
  INFINITE_SCROLL: `
REPEAT (SCROLL DOWN 800, \`document.querySelector("[class*='load-more'],[class*='show-more']")\`)
WAIT 2
EVAL \`window.scrollTo(0, document.body.scrollHeight)\`
WAIT 1
`.trim(),
} as const;
