import { apiRequest } from './common.ts' // ============================================================================ // Types // ============================================================================ /** Attribute-level counts across every test case run against a version. */ export interface VersionAttributeStats { total: number good: number bad: number pending: number } export interface VersionTestCaseCounts { total: number completed: number pendingReview: number partiallyReviewed: number } export interface VersionStats { promptVersionId: string status: string /** Counts over all attributes. */ stats: VersionAttributeStats /** Counts restricted to test cases where review has started. */ reviewedStats: Omit testCases: VersionTestCaseCounts } export interface VersionStatsFilters { createdAtSince?: string createdAtUntil?: string updatedAtSince?: string updatedAtUntil?: string createdBy?: string[] testCaseTitle?: string status?: string[] tagIds?: string[] } /** Review scoreboard for a version: how many attributes carry a verdict. */ export interface GenerationStats { positive: number negative: number unlabeled: number } export interface VersionComparisonRow { promptVersionId: string title?: string stats: VersionAttributeStats testCases: VersionTestCaseCounts /** good / (good + bad), or null when nothing has been reviewed. */ score: number | null /** Share of the version's attributes that carry a verdict, 0..1. */ coverage: number /** Test cases that produced no reviewable generation — never run, still running, or failed. */ unrunCases: number } export interface VersionComparison { rows: VersionComparisonRow[] baseline: VersionComparisonRow | null candidate: VersionComparisonRow | null /** candidate.score - baseline.score, or null when either side is not fully reviewed. */ scoreDelta: number | null /** Attributes the candidate still has not been judged on. */ pending: number /** Attributes the baseline still has not been judged on. */ baselinePending: number verdict: 'better' | 'worse' | 'same' | 'inconclusive' } // ============================================================================ // Raw stats // ============================================================================ function buildStatsQuery(filters?: VersionStatsFilters): string { const params = new URLSearchParams() if (filters?.createdAtSince) params.set('createdAtSince', filters.createdAtSince) if (filters?.createdAtUntil) params.set('createdAtUntil', filters.createdAtUntil) if (filters?.updatedAtSince) params.set('updatedAtSince', filters.updatedAtSince) if (filters?.updatedAtUntil) params.set('updatedAtUntil', filters.updatedAtUntil) if (filters?.testCaseTitle) params.set('testCaseTitle', filters.testCaseTitle) for (const value of filters?.createdBy ?? []) params.append('createdBy', value) for (const value of filters?.status ?? []) params.append('status', value) for (const value of filters?.tagIds ?? []) params.append('tagIds', value) const query = params.toString() return query ? `?${query}` : '' } /** * The scoreboard for one prompt version: how many output attributes were judged good, bad, or are * still pending, plus the test case counts behind them. * * This is the measurement half of the prompt feedback loop — the answer to "did the change help". * Works for canvas and workflow versions alike. */ export async function getVersionStats( promptVersionId: string, filters?: VersionStatsFilters, ): Promise { return apiRequest(`/prompt-version/${promptVersionId}/stats${buildStatsQuery(filters)}`) } /** * Counts of the **overall** thumb on each generation (`generation.feedback`) for a version. * * This is not the per-attribute scoreboard — that is {@link getVersionStats}. Nothing in this skill * writes `generation.feedback`, so the two will usually disagree. */ export async function getGenerationStats( promptVersionId: string, options?: { hasTestCaseId?: boolean }, ): Promise { const params = new URLSearchParams({ promptVersionId }) if (options?.hasTestCaseId !== undefined) params.set('hasTestCaseId', String(options.hasTestCaseId)) return apiRequest(`/generation/stats?${params.toString()}`) } // ============================================================================ // Comparison // ============================================================================ function scoreOf(stats: VersionAttributeStats): number | null { const judged = stats.good + stats.bad return judged ? stats.good / judged : null } function coverageOf(stats: VersionAttributeStats): number { return stats.total ? (stats.good + stats.bad) / stats.total : 0 } /** * Test cases that produced no reviewable generation — never run, still running, or failed. * * They contribute no attributes, so they are invisible in `stats.pending`: a suite where two of five * cases never ran reads as fully reviewed. Count them separately. */ function unrunCasesOf(testCases: VersionTestCaseCounts): number { const accounted = testCases.completed + testCases.pendingReview + testCases.partiallyReviewed return Math.max(0, testCases.total - accounted) } /** * Score two or more versions side by side. * * The stats endpoint answers for one version at a time — passing several ids returns a single * aggregate, not a per-version breakdown — so this fetches each and assembles the table itself. * * Order matters: the **first** id is the baseline and the **last** is the candidate. * * `verdict` is `inconclusive` unless **both** sides are fully reviewed. A score computed over a * partially reviewed version describes only the attributes someone happened to look at: one good * result and ninety-nine unjudged ones scores 100%, which is not a fact about quality. Read * `coverage` on each row to see how much of it was actually reviewed. */ export async function compareVersions( promptVersionIds: string[], options?: { filters?: VersionStatsFilters, titles?: Record }, ): Promise { if (promptVersionIds.length < 2) throw new Error('compareVersions needs at least two version ids: the baseline first, the candidate last.') const results = await Promise.all( promptVersionIds.map(id => getVersionStats(id, options?.filters)), ) const rows: VersionComparisonRow[] = results.map((result, index) => ({ promptVersionId: promptVersionIds[index]!, ...(options?.titles?.[promptVersionIds[index]!] && { title: options.titles[promptVersionIds[index]!] }), stats: result.stats, testCases: result.testCases, score: scoreOf(result.stats), coverage: coverageOf(result.stats), unrunCases: unrunCasesOf(result.testCases), })) const baseline = rows[0]! const candidate = rows[rows.length - 1]! // An unrun case contributes no attributes, so `pending === 0` alone would call a half-executed // suite fully reviewed. const fullyReviewed = baseline.stats.pending === 0 && candidate.stats.pending === 0 && baseline.unrunCases === 0 && candidate.unrunCases === 0 const scoreDelta = fullyReviewed && baseline.score !== null && candidate.score !== null ? candidate.score - baseline.score : null let verdict: VersionComparison['verdict'] = 'inconclusive' if (scoreDelta !== null) verdict = scoreDelta > 0 ? 'better' : scoreDelta < 0 ? 'worse' : 'same' return { rows, baseline, candidate, scoreDelta, pending: candidate.stats.pending, baselinePending: baseline.stats.pending, verdict, } } /** * Render a comparison as a table for the user. * * An inconclusive verdict is spelled out with the reason, because "inconclusive" on its own reads * as a tool failure rather than as work still to do. */ export function formatVersionComparison(comparison: VersionComparison): string { const lines = ['version good bad pending score'] for (const row of comparison.rows) { const label = (row.title ?? row.promptVersionId).slice(0, 36).padEnd(36) const score = row.score === null ? ' \u2014 ' : `${(row.score * 100).toFixed(0)}%`.padStart(5) lines.push( `${label}${String(row.stats.good).padStart(4)}${String(row.stats.bad).padStart(5)}${String(row.stats.pending).padStart(9)} ${score}`, ) } lines.push('') if (comparison.scoreDelta !== null) { const sign = comparison.scoreDelta >= 0 ? '+' : '' lines.push(`verdict: ${comparison.verdict} (${sign}${(comparison.scoreDelta * 100).toFixed(0)} points)`) return lines.join('\n') } const unrun = comparison.rows.reduce((total, row) => total + row.unrunCases, 0) if (unrun) { lines.push( `verdict: inconclusive \u2014 ${unrun} test case(s) produced no result, so the suite is ` + 'only partly executed. Run it in full, review it, then compare again.', ) return lines.join('\n') } const unreviewed: string[] = [] if (comparison.baselinePending) unreviewed.push(`${comparison.baselinePending} attribute(s) on the baseline`) if (comparison.pending) unreviewed.push(`${comparison.pending} attribute(s) on the candidate`) lines.push( unreviewed.length ? `verdict: inconclusive \u2014 ${unreviewed.join(' and ')} are still unreviewed. ` + 'A score over a partial review is not a measurement; finish reviewing, then compare again.' : 'verdict: inconclusive \u2014 neither version has any judged attributes.', ) return lines.join('\n') } // ============================================================================ // Diagnosis — what was actually sent to the model // ============================================================================ export interface CompletionRun { id: string status: 'created' | 'running' | 'succeeded' | 'failed' | (string & {}) /** The variables as they were resolved for the call, files included. */ rawInput: unknown /** The provider's response, before Tela post-processing. */ rawOutput: unknown inputContent: unknown outputContent: unknown /** Carries `promptVersion.modelConfigurations` — the model, temperature and schema actually used. */ metadata: Record | null creditsUsed: number tags: string[] promptId: string | null promptVersionId: string | null promptApplicationId: string | null createdAt: string updatedAt: string } export interface ListCompletionRunsOptions { promptId?: string promptVersionId?: string promptApplicationId?: string tags?: string limit?: number offset?: number } /** * A single canvas execution, with the resolved input, the raw model output, the model * configuration it ran under, and what it cost. * * Note the route is under `/v1/`, unlike most of the API. */ export async function getCompletionRun(completionRunId: string): Promise { return apiRequest(`/v1/completion-run/${completionRunId}`) } /** * Canvas executions for a prompt, version, or application, newest first. * * Covers **Workstation tasks and direct API completions only**. Test case runs do not create * completion runs, so a canvas exercised solely through its test suite reports zero here — which * means the resolved prompt behind a test case result is not retrievable through the API. * * **Use this for diagnosis, never for cost.** `creditsUsed` and `usage.cost` on a run are * execution-time artifacts: not reconciled, not what the workspace is billed. Cost comes from the * usage service — see `usage.ts`. * * A generation cannot be joined to a run by id either: `generation.metadata.executionId` is a * Trigger.dev run id, not a completion run. Match on `promptVersionId` and timestamp. */ export async function listCompletionRuns( options?: ListCompletionRunsOptions, ): Promise<{ data: CompletionRun[], meta: { totalCount: number, limit: number, offset: number } }> { const params = new URLSearchParams() if (options?.promptId) params.set('promptId', options.promptId) if (options?.promptVersionId) params.set('promptVersionId', options.promptVersionId) if (options?.promptApplicationId) params.set('promptApplicationId', options.promptApplicationId) if (options?.tags) params.set('tags', options.tags) if (options?.limit !== undefined) params.set('limit', String(options.limit)) if (options?.offset !== undefined) params.set('offset', String(options.offset)) const query = params.toString() return apiRequest(`/v1/completion-run${query ? `?${query}` : ''}`) } // ============================================================================ // Readiness — is this version fit to ship // ============================================================================ export interface VersionReadiness { ready: boolean /** Test cases that produced no reviewable generation — never run, still running, or failed. */ unrunCases: number /** good / (good + bad) over the reviewed attributes, or null when nothing was reviewed. */ score: number | null reviewed: number pending: number failing: number /** Reasons it is not ready. Empty when `ready` is true. */ blockers: string[] } /** * Whether a version is fit to promote, or to point a Workstation at. * * Running production tasks on an unreviewed version spends real money on output nobody has checked. * Call this before promoting and before creating tasks in bulk, and report the blockers rather than * proceeding. * * `minScore` defaults to 1: every reviewed attribute must be good. Lower it deliberately, and say * what you lowered it to. */ export async function getVersionReadiness( promptVersionId: string, options: { minScore?: number } = {}, ): Promise { const minScore = options.minScore ?? 1 const { stats, testCases } = await getVersionStats(promptVersionId) const reviewed = stats.good + stats.bad const score = reviewed ? stats.good / reviewed : null const unrunCases = unrunCasesOf(testCases) const blockers: string[] = [] // `stats` is computed over generations, so it is empty both before the suite runs and when the // prompt declares no attributes. `testCases.completed` does not disambiguate — it counts test // cases that are *done being reviewed*, not ones that have executed — so name both causes rather // than guessing. Report only the first blocker that applies, so there is one next step. if (!testCases.total) { blockers.push('the version has no test cases — there is nothing to judge it by') } else if (!stats.total) { blockers.push( `no reviewable attributes on this version — either the suite has not run against it yet ` + `(${testCases.total} test case(s) exist), or the prompt declares no structuredOutput`, ) } else if (stats.pending === stats.total) { blockers.push(`none of the ${stats.total} attribute(s) have been reviewed`) } else if (stats.pending) { blockers.push(`${stats.pending} of ${stats.total} attribute(s) are still unreviewed`) } // An unrun case contributes no attributes, so it is invisible in `stats.pending`: three of five // cases passing would otherwise read as a fully reviewed suite. if (unrunCases) { blockers.push( `${unrunCases} of ${testCases.total} test case(s) produced no result — never run, still ` + 'running, or failed', ) } if (score !== null && score < minScore) { blockers.push( `score is ${(score * 100).toFixed(0)}% against a ${(minScore * 100).toFixed(0)}% bar ` + `(${stats.bad} attribute(s) judged bad)`, ) } return { ready: !blockers.length, unrunCases, score, reviewed, pending: stats.pending, failing: stats.bad, blockers, } } /** * Render readiness as something to show the user. */ export function formatVersionReadiness(readiness: VersionReadiness): string { if (readiness.ready) { return `Ready to ship: ${readiness.reviewed} attribute(s) reviewed, ` + `${readiness.score === null ? 'n/a' : `${(readiness.score * 100).toFixed(0)}%`} good, nothing pending.` } return ['Not ready to ship:', ...readiness.blockers.map(blocker => `- ${blocker}`)].join('\n') }