import { ApiRequestError, apiRequest } from './common.ts' // ============================================================================ // Types // ============================================================================ /** * A single named input for an agent test case. * Text inputs carry inline content; file inputs reference a Vault file. */ export type AgentTestInput = | { type: 'text', name: string, content: string } | { type: 'file', name: string, vaultRef: string, filename: string, metadata?: string } export interface AgentTestCaseValidationReferenceFile { index: number name: string mimeType?: string | null vaultRef: string url?: string } export interface AgentTestCaseAnswers { good: unknown[] bad: unknown[] } export interface AgentTestCaseTag { testCaseTagAssignmentId?: string id: string name: string color: string promptId: string | null agentId: string | null workspaceId: string createdAt: string updatedAt: string deletedAt: string | null } export interface AgentTestCase { id: string agentId: string workspaceId: string title: string inputs: AgentTestInput[] expectedOutput: Record | string | null validationReferenceFiles: AgentTestCaseValidationReferenceFile[] evaluationInstructions: string | null answers: Record metadata: Record tags: AgentTestCaseTag[] createdBy: string createdAt: string updatedAt: string } export type AgentTestCaseFeedback = 0 | 1 | null export type AgentTestCaseValidationType = 'good' | 'bad' | 'warning' | 'missing' export interface AgentTestCaseValidationSummaryEntry { type: AgentTestCaseValidationType source?: 'llm-judge' | 'manual' | 'answer-match' message?: string matchedAnswer?: unknown } export interface AgentTestCaseRun { id: string testId: string commitHash: string executionSessionId: string | null executionStatus: 'pending' | 'running' | 'completed' | 'error' | null executionError: string | null savedInputs: AgentTestInput[] savedChatMessages: Array<{ content: string stepEndIndex: number attachments?: Array<{ vaultRef: string, fileName: string, fileType: string }> output?: unknown duration?: number }> output: Record | string | null executionUsage: Record | null evaluationSessionId: string | null evaluationInputSnapshot: unknown evaluationRequirements: string evaluationStatus: 'pending' | 'running' | 'completed' | 'failed' | 'error' | null evaluationError: string | null feedback: AgentTestCaseFeedback attributeFeedback: Record validationSummary: Record evaluatedAt: string | null createdAt: string updatedAt: string } /** * Derived run status — not a stored column. Computed from execution/evaluation * state by getAgentTestCaseRunStatus. Terminal: success, failed, error. */ export type AgentTestCaseStatus = 'new' | 'running' | 'executed' | 'success' | 'failed' | 'error' export interface AgentTestCaseCounters { total: number good: number bad: number pending: number } export interface AgentTestMetric { id: string agentId: string workspaceId: string name: string attributeKeys: string[] position: number createdBy: string updatedBy: string | null createdAt: string updatedAt: string } export type AgentTestMetricStatus = 'completed' | 'pending' | 'partial' | 'unavailable' export interface AgentTestMetricStats { id: string name: string attributeKeys: string[] availableAttributeKeys: string[] missingAttributeKeys: string[] stats: AgentTestCaseCounters reviewedStats?: { total: number, good: number, bad: number } score: number | null status: AgentTestMetricStatus } export interface AgentTestCaseWithRun { test: AgentTestCase run: AgentTestCaseRun | null stats?: AgentTestCaseCounters metricStats?: AgentTestMetricStats[] } export interface AgentTestCaseStats { agentId: string commitHash: string status: 'pending' | 'completed' stats: AgentTestCaseCounters reviewedStats: { total: number, good: number, bad: number } testCases: { total: number executed: number evaluated: number passed: number failed: number pendingExecution: number pendingEvaluation: number running: number error: number } metrics: AgentTestMetricStats[] } export interface AgentTestCaseRunHistoryEntry { commitHash: string createdAt: string score: number total: number good: number bad: number pending: number attributeOutcomes: Record } export interface ListAgentTestCaseRunsResult { data: AgentTestCaseRunHistoryEntry[] nextCursor: string | null totalRuns: number } export interface AgentCommit { commitHash: string shortHash: string message: string createdAt: string author: { name: string, email: string, avatarUrl?: string } } export interface AgentHistory { commits: AgentCommit[] branch: string pagination: { page: number, limit: number, totalCount: number, hasMore: boolean } } // ============================================================================ // Payloads & Options // ============================================================================ export interface CreateAgentTestCasePayload { title?: string inputs?: AgentTestInput[] expectedOutput?: Record | string | null validationReferenceFiles?: AgentTestCaseValidationReferenceFile[] evaluationInstructions?: string | null answers?: Record metadata?: Record } export type UpdateAgentTestCasePayload = CreateAgentTestCasePayload export interface ListAgentTestCasesOptions { /** Attach the run for this agent commit to each test case (enables stats) */ commitHash?: string /** Search by test case title */ title?: string /** Filter by creator user IDs */ createdBy?: string[] createdAtSince?: string createdAtUntil?: string updatedAtSince?: string updatedAtUntil?: string /** Filter by derived run status (requires commitHash) */ status?: AgentTestCaseStatus[] /** Filter by tag IDs */ tagIds?: string[] } export interface RunAllAgentTestCasesPayload { commitHash: string /** Explicit test case IDs — mutually exclusive with filters */ testCaseIds?: string[] /** Filter-based selection — mutually exclusive with testCaseIds */ filters?: { title?: string createdBy?: string[] createdAtSince?: string createdAtUntil?: string updatedAtSince?: string updatedAtUntil?: string tagIds?: string[] } } export type AgentTestCaseSkipReason = | 'already-running' | 'already-passed' | 'already-failed' | 'missing-input' | 'unknown-input' | 'invalid-input' | 'not-accessible' | 'usage-limit' export interface RunAllAgentTestCasesResult { executionId: string executionIds: string[] testCaseIds: string[] skipped: Array<{ testCaseId: string, reason: AgentTestCaseSkipReason }> } export interface AgentTestCaseAttributeFeedback { path: string feedback: AgentTestCaseFeedback type?: AgentTestCaseValidationType message?: string } export interface CreateAgentTestMetricPayload { name: string attributeKeys: string[] position?: number } export type UpdateAgentTestMetricPayload = Partial // ============================================================================ // Internal helpers // ============================================================================ // Agent test case endpoints use `vaultRef` fields that must NOT be rewritten // by parsePayload's vault:// → { file_url } expansion, so every body is // pre-serialized with JSON.stringify (string bodies skip parsePayload). function jsonBody(payload: unknown): string { return JSON.stringify(payload) } function buildListQuery(options?: ListAgentTestCasesOptions): string { const params = new URLSearchParams() if (options?.commitHash) params.set('commitHash', options.commitHash) if (options?.title) params.set('title', options.title) if (options?.createdAtSince) params.set('createdAtSince', options.createdAtSince) if (options?.createdAtUntil) params.set('createdAtUntil', options.createdAtUntil) if (options?.updatedAtSince) params.set('updatedAtSince', options.updatedAtSince) if (options?.updatedAtUntil) params.set('updatedAtUntil', options.updatedAtUntil) // Arrays use repeated keys (Elysia query array format) for (const v of options?.createdBy ?? []) params.append('createdBy', v) for (const v of options?.status ?? []) params.append('status', v) for (const v of options?.tagIds ?? []) params.append('tagIds', v) const qs = params.toString() return qs ? `?${qs}` : '' } // ============================================================================ // Test Cases (CRUD) // ============================================================================ /** * List agent test cases with optional filters. * Pass commitHash to attach that commit's run (and stats) to each test case. * Returns all matching cases (no pagination on this endpoint). */ export async function listAgentTestCases( agentId: string, options?: ListAgentTestCasesOptions, ): Promise { const { data } = await apiRequest<{ data: AgentTestCaseWithRun[] }>( `/agent/${agentId}/tests${buildListQuery(options)}`, ) return data } /** * Create a new test case for an agent. * Use createAgentTestCasePayload/buildAgentTestInputs to build inputs. */ export async function createAgentTestCase( agentId: string, payload: CreateAgentTestCasePayload = {}, ): Promise { const { data } = await apiRequest<{ data: AgentTestCase }>(`/agent/${agentId}/tests`, { method: 'POST', body: jsonBody(payload), }) return data } /** * Update an existing agent test case (title, inputs, expectations, etc.) */ export async function updateAgentTestCase( agentId: string, testId: string, payload: UpdateAgentTestCasePayload, ): Promise { const { data } = await apiRequest<{ data: AgentTestCase }>(`/agent/${agentId}/tests/${testId}`, { method: 'PATCH', body: jsonBody(payload), }) return data } /** * Delete an agent test case (soft delete) */ export async function deleteAgentTestCase(agentId: string, testId: string): Promise { await apiRequest(`/agent/${agentId}/tests/${testId}`, { method: 'DELETE' }) } // ============================================================================ // Execution & Evaluation // ============================================================================ /** * Run a test case against an agent version (commitHash). * Returns 202 immediately — poll getAgentTestCaseRun / waitForAgentTestCaseRun. * One run exists per (testId, commitHash); re-running replaces it. */ export async function runAgentTestCase( agentId: string, testId: string, commitHash: string, ): Promise<{ executionId: string }> { const { data } = await apiRequest<{ data: { executionId: string } }>( `/agent/${agentId}/tests/${testId}/run`, { method: 'POST', body: jsonBody({ commitHash }) }, ) return data } /** * Continue a test case session with a new user message (multiturn). * Requires an existing, non-running run for the commit. Re-evaluates after the turn. */ export async function continueAgentTestCase( agentId: string, testId: string, commitHash: string, message: string, ): Promise<{ executionId: string }> { const { data } = await apiRequest<{ data: { executionId: string } }>( `/agent/${agentId}/tests/${testId}/continue`, { method: 'POST', body: jsonBody({ commitHash, message }) }, ) return data } /** * Run all (or a subset of) test cases against an agent version in batch. * Select via explicit testCaseIds OR filters (mutually exclusive). * Already-passed/failed/running cases are skipped with a reason. */ export async function runAllAgentTestCases( agentId: string, payload: RunAllAgentTestCasesPayload, ): Promise { const { data } = await apiRequest<{ data: RunAllAgentTestCasesResult }>( `/agent/${agentId}/tests/run-all`, { method: 'POST', body: jsonBody(payload) }, ) return data } /** * Get the run of a test case for a specific agent commit */ export async function getAgentTestCaseRun( agentId: string, testId: string, commitHash: string, ): Promise { const { data } = await apiRequest<{ data: AgentTestCaseRun }>( `/agent/${agentId}/tests/${testId}/runs/${commitHash}`, ) return data } /** * List a test case's run history across commits (keyset pagination) */ export async function listAgentTestCaseRuns( agentId: string, testId: string, options?: { limit?: number, cursor?: string }, ): Promise { const params = new URLSearchParams() if (options?.limit !== undefined) params.set('limit', String(options.limit)) if (options?.cursor) params.set('cursor', options.cursor) const qs = params.toString() // Unlike the other endpoints, pagination fields live on the envelope itself: // the API returns { data: entries, nextCursor, totalRuns } return await apiRequest( `/agent/${agentId}/tests/${testId}/runs${qs ? `?${qs}` : ''}`, ) } /** * Record manual thumbs up/down per output attribute path on a run. * Also feeds the test case's answer bank (answers[path].good/bad), which * auto-classifies future runs via answer-match. feedback: 1=good, 0=bad, null=clear. */ export async function updateAgentTestCaseAttributeFeedback( agentId: string, testId: string, commitHash: string, attributes: AgentTestCaseAttributeFeedback[], ): Promise { const { data } = await apiRequest<{ data: AgentTestCaseRun }>( `/agent/${agentId}/tests/${testId}/runs/${commitHash}/attribute-feedback`, { method: 'PATCH', body: jsonBody({ attributes }) }, ) return data } // ============================================================================ // Stats & Metrics // ============================================================================ /** * Aggregated test case stats for an agent commit (scores per output attribute, * counters, custom metrics). Accepts the same filters as listAgentTestCases. */ export async function getAgentTestCaseStats( agentId: string, commitHash: string, options?: Omit, ): Promise { const { data } = await apiRequest<{ data: AgentTestCaseStats }>( `/agent/${agentId}/tests/stats${buildListQuery({ ...options, commitHash })}`, ) return data } /** * List custom test metrics for an agent */ export async function listAgentTestMetrics(agentId: string): Promise { const { data } = await apiRequest<{ data: AgentTestMetric[] }>(`/agent/${agentId}/test-metrics`) return data } /** * Create a custom test metric grouping output attributes. * attributeKeys must exist in the agent's output-format.json. */ export async function createAgentTestMetric( agentId: string, payload: CreateAgentTestMetricPayload, ): Promise { const { data } = await apiRequest<{ data: AgentTestMetric }>(`/agent/${agentId}/test-metrics`, { method: 'POST', body: jsonBody(payload), }) return data } /** * Update a custom test metric */ export async function updateAgentTestMetric( agentId: string, metricId: string, payload: UpdateAgentTestMetricPayload, ): Promise { const { data } = await apiRequest<{ data: AgentTestMetric }>( `/agent/${agentId}/test-metrics/${metricId}`, { method: 'PATCH', body: jsonBody(payload) }, ) return data } /** * Delete a custom test metric (soft delete) */ export async function deleteAgentTestMetric(agentId: string, metricId: string): Promise { await apiRequest(`/agent/${agentId}/test-metrics/${metricId}`, { method: 'DELETE' }) } // ============================================================================ // Tags // ============================================================================ /** * List an agent's test case tags */ export async function listAgentTestCaseTags(agentId: string): Promise { const { data } = await apiRequest<{ data: AgentTestCaseTag[] }>(`/agent/${agentId}/tags`) return data } /** * Create a tag scoped to an agent */ export async function createAgentTestCaseTag( agentId: string, payload: { name: string, color: string }, ): Promise { const { data } = await apiRequest<{ data: AgentTestCaseTag }>(`/agent/${agentId}/tags`, { method: 'POST', body: jsonBody(payload), }) return data } /** * Update a tag's name/color */ export async function updateAgentTestCaseTag( agentId: string, tagId: string, payload: { name?: string, color?: string }, ): Promise { const { data } = await apiRequest<{ data: AgentTestCaseTag }>(`/agent/${agentId}/tags/${tagId}`, { method: 'PATCH', body: jsonBody(payload), }) return data } /** * Delete a tag (removes all its assignments) */ export async function deleteAgentTestCaseTag(agentId: string, tagId: string): Promise { await apiRequest(`/agent/${agentId}/tags/${tagId}`, { method: 'DELETE' }) } /** * List tags assigned to a test case */ export async function getAgentTestCaseTags( agentId: string, testId: string, ): Promise { const { data } = await apiRequest<{ data: AgentTestCaseTag[] }>( `/agent/${agentId}/tests/${testId}/tags`, ) return data } /** * Assign tags to a test case */ export async function addAgentTestCaseTags( agentId: string, testId: string, tagIds: string[], ): Promise { await apiRequest(`/agent/${agentId}/tests/${testId}/tags`, { method: 'POST', body: jsonBody({ tagIds }), }) } /** * Remove one tag assignment from a test case */ export async function removeAgentTestCaseTag( agentId: string, testId: string, tagId: string, ): Promise { await apiRequest(`/agent/${agentId}/tests/${testId}/tags/${tagId}`, { method: 'DELETE' }) } /** * Assign tags to many test cases at once */ export async function bulkAddAgentTestCaseTags( agentId: string, testCaseIds: string[], tagIds: string[], ): Promise { await apiRequest(`/agent/${agentId}/tests/bulk-add-tags`, { method: 'POST', body: jsonBody({ testCaseIds, tagIds }), }) } /** * Remove tags from many test cases at once */ export async function bulkRemoveAgentTestCaseTags( agentId: string, testCaseIds: string[], tagIds: string[], ): Promise { await apiRequest(`/agent/${agentId}/tests/bulk-remove-tags`, { method: 'POST', body: jsonBody({ testCaseIds, tagIds }), }) } // ============================================================================ // Agent version helpers // ============================================================================ /** * Get an agent's commit history (agent versions) */ export async function getAgentHistory( agentId: string, options?: { branch?: string, page?: number, limit?: number }, ): Promise { const params = new URLSearchParams() if (options?.branch) params.set('branch', options.branch) if (options?.page !== undefined) params.set('page', String(options.page)) if (options?.limit !== undefined) params.set('limit', String(options.limit)) const qs = params.toString() const { data } = await apiRequest<{ data: AgentHistory }>( `/agent/${agentId}/history${qs ? `?${qs}` : ''}`, ) return data } /** * Resolve the agent's latest commitHash — the version test cases run against */ export async function getLatestAgentCommit(agentId: string): Promise { const history = await getAgentHistory(agentId, { limit: 1 }) const commitHash = history.commits[0]?.commitHash if (!commitHash) { throw new Error(`Agent ${agentId} has no commits`) } return commitHash } // ============================================================================ // Helper Functions // ============================================================================ /** * Build the inputs array from a simple Record of variable values. * Values starting with vault:// become file inputs; everything else is text. * fileNames maps input names to display filenames (defaults to the input name). */ export function buildAgentTestInputs( variables: Record, options?: { fileNames?: Record }, ): AgentTestInput[] { return Object.entries(variables).map(([name, value]) => { if (typeof value === 'string' && value.startsWith('vault://')) { return { type: 'file' as const, name, vaultRef: value, filename: options?.fileNames?.[name] ?? name, } } return { type: 'text' as const, name, content: value } }) } /** * Create an agent test case payload from variable values. * The evaluator needs at least one expectation source to run: * expectedOutput, validationReferenceFiles, or evaluationInstructions. */ export function createAgentTestCasePayload( variables: Record, options?: { title?: string expectedOutput?: Record | string evaluationInstructions?: string validationReferenceFiles?: AgentTestCaseValidationReferenceFile[] metadata?: Record fileNames?: Record }, ): CreateAgentTestCasePayload { const inputs = buildAgentTestInputs(variables, { fileNames: options?.fileNames }) const firstText = inputs.find(i => i.type === 'text' && i.content) const firstFile = inputs.find(i => i.type === 'file') const fallbackTitle = firstText?.type === 'text' ? `Test: ${firstText.content.slice(0, 30)}...` : firstFile?.type === 'file' ? `Test: ${firstFile.filename}` : 'Untitled Test' return { title: options?.title ?? fallbackTitle, inputs, expectedOutput: options?.expectedOutput, evaluationInstructions: options?.evaluationInstructions, validationReferenceFiles: options?.validationReferenceFiles, metadata: options?.metadata, } } /** * Derive the run status. Intentionally a verbatim copy of the backend's * getAgentTestCaseRunStatus (agent-test-case.controller.ts) so client-side * status always matches API filtering and the UI — including completed * evaluations without positive feedback counting as failed. * Terminal: success, failed, error. `executed` = execution done, not evaluated. */ export function getAgentTestCaseRunStatus( run: Pick< AgentTestCaseRun, 'executionSessionId' | 'executionStatus' | 'evaluationSessionId' | 'evaluationStatus' | 'feedback' > | null | undefined, ): AgentTestCaseStatus { if (!run) return 'new' if (run.executionStatus === 'error' || run.evaluationStatus === 'error') return 'error' if ( run.executionStatus === 'pending' || run.executionStatus === 'running' || (run.evaluationSessionId && (run.evaluationStatus === 'pending' || run.evaluationStatus === 'running')) ) { return 'running' } if (run.executionStatus === 'completed' && !run.evaluationSessionId) return 'executed' if (run.evaluationStatus === 'completed') return run.feedback === 1 ? 'success' : 'failed' if (run.evaluationStatus === 'failed' || run.feedback === 0) return 'failed' if (run.executionSessionId) return 'executed' return 'new' } /** * Poll a test case run until it reaches a terminal status. * The run row may not exist yet right after triggering — 404s are retried. * By default waits for evaluation (success/failed/error); pass * until: 'executed' to stop as soon as execution completes. */ export async function waitForAgentTestCaseRun( agentId: string, testId: string, commitHash: string, options?: { intervalMs?: number, maxAttempts?: number, until?: 'evaluated' | 'executed' }, ): Promise<{ run: AgentTestCaseRun, status: AgentTestCaseStatus }> { const interval = options?.intervalMs ?? 2000 const maxAttempts = options?.maxAttempts ?? 300 // 10 minutes default const until = options?.until ?? 'evaluated' for (let i = 0; i < maxAttempts; i++) { let run: AgentTestCaseRun | null = null try { run = await getAgentTestCaseRun(agentId, testId, commitHash) } catch (error) { // Run row is created asynchronously by the trigger task if (!(error instanceof ApiRequestError) || error.status !== 404) { throw error } } if (run) { const status = getAgentTestCaseRunStatus(run) if (status === 'success' || status === 'failed' || status === 'error') { return { run, status } } if (status === 'executed' && until === 'executed') { return { run, status } } } await new Promise(resolve => setTimeout(resolve, interval)) } throw new Error(`Agent test case run ${testId}@${commitHash} did not complete within timeout`) }