import type { PhoenixClient } from "../index.js";
import type { AnnotatorKind } from "../types/annotations.js";
import type { EvaluatorParams, EvaluationResult as ExperimentEvaluationResult } from "../types/experiments.js";
/**
 * Phoenix annotator kind, re-exported from the shared client types so the
 * testing module and the rest of the client agree on a single definition.
 */
export type { AnnotatorKind };
/** A JSON-serializable map. */
export type KVMap = Record<string, unknown>;
/**
 * Domain language
 * ---------------
 * The unit these tests evaluate over is an **Example**: a single AI example —
 * an `input`, its `expected` output, and optional `metadata` / `splits` — over
 * which the task under test is run and then scored. This is the same notion as
 * the dataset `Example` (`../types/datasets`): each test case _is_ one example.
 * When tracked, a case is recorded to Phoenix as a dataset example and
 * evaluated as one experiment run.
 *
 * These field names form the shared vocabulary across this module:
 * - `input`    — the example's input, passed to the task under evaluation.
 * - `expected` — the example's expected (reference / ground-truth) output.
 * - `metadata` — extra fields carried on the example.
 * - `splits`   — slice labels for the example.
 * - `id`       — stable example id, used to upsert the example across runs.
 */
/**
 * The expected output of an `Example`, accepted under any one of three
 * interchangeable keys. All three normalize to the same slot: when recorded to
 * Phoenix the value becomes the dataset example's `output`, and it is exposed
 * to evaluators as `expected` on `EvaluatorParams`. At most one key may be set.
 *
 * - `expected` — the canonical name (the ground-truth / reference output).
 * - `reference` — alias preferred by frameworks that name the slot "reference".
 * - `output` — alias for callers who think in terms of the example's `output`.
 *
 * Modeled as a union so supplying more than one key at a time is a type error.
 */
export type ReferenceOutput<Expected extends KVMap = KVMap> = {
    expected?: Expected;
    reference?: never;
    output?: never;
} | {
    reference?: Expected;
    expected?: never;
    output?: never;
} | {
    output?: Expected;
    expected?: never;
    reference?: never;
};
/**
 * The `Example` fields that define a single test case, excluding its
 * expected output (which is supplied separately via {@link ReferenceOutput}).
 *
 * `input` is the example's input — the value fed to the task under evaluation.
 * When the case is tracked, this becomes the dataset example's `input`.
 */
export interface TestParamsBase<Input extends KVMap = KVMap> {
    /** Optional stable example id; used to upsert the example between runs. */
    id?: string;
    /** The example's input — fed to the task under evaluation. Required. */
    input: Input;
    /** Additional metadata stored on the example and its run. */
    metadata?: KVMap;
    /**
     * Split assignment(s) for the example, used to slice the dataset and
     * experiment in the Phoenix UI (e.g. `["factual_accuracy", "correct"]`).
     */
    splits?: string[];
    /** Per-test config (tags + metadata recorded on the run). */
    config?: TestConfig;
    /**
     * Number of times to run this test case. Each repetition becomes a
     * separate experiment run against the same dataset example (carrying a
     * distinct `repetition_number`). Overrides the suite-level `repetitions`.
     * Defaults to the suite value, then `PHOENIX_TEST_REPETITIONS`, then `1`.
     */
    repetitions?: number;
    /**
     * When `true`, this test runs as an ordinary local test only — no dataset
     * example is created and no experiment run or annotations are uploaded to
     * Phoenix. Useful for scaffolding a case before it's ready to track.
     */
    dryRun?: boolean;
}
/**
 * The full inline definition of a single `Example` under test.
 *
 * Combines {@link TestParamsBase} with a {@link ReferenceOutput}, so the
 * example's expected output may be given under `expected`, `reference`, or
 * `output` (at most one). All three resolve to the same canonical `expected`
 * slot.
 */
export type TestParams<Input extends KVMap = KVMap, Expected extends KVMap = KVMap> = TestParamsBase<Input> & ReferenceOutput<Expected>;
/**
 * Resolve an `Example`'s expected output from a value that may carry it
 * under any of the `expected` / `reference` / `output` aliases (see
 * {@link ReferenceOutput}). Returns the first one set, or `undefined` if none.
 */
export declare function resolveReference<Expected extends KVMap = KVMap>(params: ReferenceOutput<Expected>): Expected | undefined;
/** Per-test runtime configuration. */
export interface TestConfig {
    /** Tags recorded on the experiment run for filtering in the Phoenix UI. */
    tags?: string[];
    /** Extra metadata recorded on the experiment run. */
    metadata?: KVMap;
}
/**
 * How a criterion aggregates an annotation's scores to gate the suite:
 *
 * - `"average"` — gate on overall quality: the **mean** score across all runs
 *   must clear the criterion's `threshold`. A few weak runs are tolerated as
 *   long as the mean holds.
 * - `"passRate"` — gate on consistency: each run **passes** when the
 *   criterion's `passFn` predicate returns `true` for its annotation, and the
 *   suite passes when the **fraction** of runs that pass is at least
 *   `minPassRate` (e.g. `minPassRate: 0.9` ⇒ 90% must pass; `1` ⇒ all).
 */
export type AcceptanceMetric = "average" | "passRate";
/**
 * Optimization direction for a criterion's scores: `"maximize"` (higher is
 * better, the default) or `"minimize"` (lower is better). Controls every
 * score comparison the criterion makes.
 */
export type OptimizationDirection = "maximize" | "minimize";
/** Fields shared by every {@link AcceptanceCriterion} variant. */
export interface AcceptanceCriterionBase {
    /** Annotation name to aggregate across completed test runs. */
    annotationName: string;
}
/**
 * Gate the suite on the **mean** score: the average across all runs must clear
 * `threshold` (compared in `direction`).
 */
export interface AverageAcceptanceCriterion extends AcceptanceCriterionBase {
    metric: "average";
    /**
     * The bar the mean score must clear, compared in `direction`. Boolean scores
     * average as `1` (`true`) / `0` (`false`).
     */
    threshold: number;
    /**
     * Optimization direction; defaults to `"maximize"`. `"maximize"` treats a
     * higher mean as better (clears when `>= threshold`); `"minimize"` treats a
     * lower mean as better (clears when `<= threshold`) — use it for cost,
     * latency, or error-rate annotations.
     */
    direction?: OptimizationDirection;
}
/**
 * Gate the suite on the **pass rate**: each run passes when `passFn` returns
 * `true` for its annotation, and the suite passes when at least `minPassRate`
 * of runs do. `passFn` decides what "passing" means, so any logic works — a
 * score bar, a score range, a label match, a metadata check, etc.
 */
export interface PassRateAcceptanceCriterion extends AcceptanceCriterionBase {
    metric: "passRate";
    /**
     * Predicate deciding whether a single run passes, given the run's last
     * {@link Annotation} for `annotationName` (its `score`, `label`,
     * `explanation`, `metadata`, …). Runs whose predicate returns `true` count
     * toward the pass rate.
     */
    passFn: (annotation: Annotation) => boolean;
    /**
     * Minimum fraction of runs (`0`–`1`) that must pass for the suite to pass —
     * e.g. `0.9` requires 90% of runs to satisfy `passFn`, `1` requires all of
     * them. The suite passes when `passRate >= minPassRate`.
     */
    minPassRate: number;
}
/**
 * One aggregate acceptance rule, evaluated once after every test in the suite
 * has run. Each criterion aggregates a single annotation's scores with one
 * {@link AcceptanceMetric} and fails the suite when the result misses its bar.
 *
 * Scoring notes shared by every metric:
 * - Boolean scores count as `1` (`true`) / `0` (`false`).
 * - If a run logs the same annotation more than once, the last one counts.
 * - Skipped tests are excluded; dry-run tests are included (they still run).
 * - A criterion whose annotation was never logged on any run fails (rather
 *   than passing vacuously) — see {@link AcceptanceResultFields.failureReason}.
 */
export type AcceptanceCriterion = AverageAcceptanceCriterion | PassRateAcceptanceCriterion;
/** The computed fields added to an {@link AcceptanceCriterion} once evaluated. */
export interface AcceptanceResultFields {
    /**
     * The aggregate the criterion gated on, or `null` when there were no runs to
     * aggregate. For `"average"` this is the mean score; for `"passRate"` it is
     * the fraction of runs that passed (so a fully-passing `"passRate"` criterion
     * reports `1`).
     */
    value: number | null;
    /** Number of runs included in the aggregate. */
    sampleCount: number;
    /** Whether the aggregate cleared the criterion. */
    passed: boolean;
    /** Human-readable failure reason for invalid or empty aggregates. */
    failureReason?: string;
}
/** Computed result for one aggregate acceptance rule. */
export type AcceptanceResult = AcceptanceCriterion & AcceptanceResultFields;
/** Suite-level configuration accepted by `describe()`. */
export interface SuiteConfig {
    /** Override the dataset / experiment name used for the suite. */
    datasetName?: string;
    /** Description for the dataset and experiment. */
    description?: string;
    /** Suite-level metadata applied to every run in this experiment. */
    metadata?: KVMap;
    /** Override the Phoenix client used for syncing this suite. */
    client?: PhoenixClient;
    /**
     * Number of times to run each test case in this suite. Individual tests
     * may override this via `TestParams.repetitions`. Defaults to the
     * `PHOENIX_TEST_REPETITIONS` env var, then `1`.
     */
    repetitions?: number;
    /**
     * When `true`, the whole suite runs as ordinary local tests — no dataset
     * is uploaded and no experiment, runs, or annotations are created in
     * Phoenix. Equivalent to `PHOENIX_TEST_TRACKING=false` scoped to this
     * suite. The reporter still prints a local summary.
     */
    dryRun?: boolean;
    /**
     * Aggregate annotation criteria that gate the suite after all tests run.
     * Each criterion fails the suite when its scores miss the configured bar
     * (see {@link AcceptanceCriterion}).
     */
    acceptanceCriteria?: AcceptanceCriterion[];
}
/**
 * Arguments passed to a `test()` body: the `Example` under test, exposed
 * as its `input`, `expected` output, and `metadata`. Read straight from the
 * test's {@link TestParams} — the runner does not transform them.
 */
export interface TestArgs<Input extends KVMap = KVMap, Expected extends KVMap = KVMap> {
    /** The example's input under test. */
    input: Input;
    /** The example's expected (reference) output, when one was supplied. */
    expected?: Expected;
    /** Any metadata attached to the example. */
    metadata?: KVMap;
}
/**
 * Object form of an evaluator result. Reuses the shared experiment
 * {@link ExperimentEvaluationResult} shape (label / explanation / metadata)
 * but widens `score` to also accept booleans, which the testing API stores as
 * `1` / `0`.
 */
export interface EvaluationResultObject extends Omit<ExperimentEvaluationResult, "score"> {
    /** Numeric or boolean score; booleans are stored as `1` / `0`. */
    score?: number | boolean | null;
}
/**
 * One annotation recorded against a run. Extends the evaluator
 * {@link EvaluationResultObject} with the `name` and `annotatorKind` carried
 * on the evaluation body, plus an optional originating trace id.
 */
export interface Annotation extends EvaluationResultObject {
    /** Phoenix evaluation name. Required, and unique per run (last write wins). */
    name: string;
    /** Who or what produced the annotation. Defaults to `"CODE"`. */
    annotatorKind?: AnnotatorKind;
    /** Trace id for this evaluation, when the annotation was produced by a traced evaluator. */
    traceId?: string | null;
}
/** Result returned by `traceEvaluator` for any evaluator-shaped value. */
export type EvaluatorResult = Annotation | (KVMap & {
    name: string;
});
/** Result shape produced by evaluator objects used in eval tests. */
export type EvaluationResult = number | boolean | string | null | EvaluationResultObject;
/**
 * Parameters passed to an evaluator when it runs inside a test. A relaxation of
 * the shared {@link EvaluatorParams}: `input` is always present, while `output`
 * (an evaluator may run before `logOutput()`) and the remaining fields are
 * optional. Deriving from `EvaluatorParams` keeps this aligned with the
 * experiment evaluator contract as that shape evolves.
 */
export type EvaluationParams = Partial<EvaluatorParams> & {
    /** The example's input under test. */
    input: KVMap;
};
/** Structural evaluator interface accepted by `evaluate()`. */
export interface Evaluator<Params extends KVMap = EvaluationParams & KVMap, Result = EvaluationResult> {
    /** Annotation/evaluation name. */
    name: string;
    /** Who or what produced the result. Defaults to `"CODE"`. */
    kind?: AnnotatorKind;
    /** Compute the evaluation result. */
    evaluate: (params: Params) => Result | Promise<Result>;
}
/** Test handler signature. */
export type TestFn<Input extends KVMap = KVMap, Expected extends KVMap = KVMap> = (args: TestArgs<Input, Expected>) => unknown | Promise<unknown>;
/**
 * Each-row shape accepted by `test.each(table)(name, fn)`; each row defines one
 * `Example`.
 *
 * Like {@link TestParams}, the example's expected output is supplied via
 * {@link ReferenceOutput} (`expected` / `reference` / `output`, at most one).
 * The trailing index signature still permits arbitrary extra columns on a row
 * (e.g. for `%j` name interpolation) without weakening that constraint.
 */
export type TestEachRow<Input extends KVMap = KVMap, Expected extends KVMap = KVMap> = {
    id?: string;
    input: Input;
    metadata?: KVMap;
    /** Per-row split assignment(s); see `TestParams.splits`. */
    splits?: string[];
    /** Per-row repetition count; see `TestParams.repetitions`. */
    repetitions?: number;
    /** Per-row dry-run flag; see `TestParams.dryRun`. */
    dryRun?: boolean;
} & ReferenceOutput<Expected> & Record<string, unknown>;
//# sourceMappingURL=types.d.ts.map