import { beforeEach, describe, expect, mock, test } from 'bun:test' const apiRequest = mock(async (_endpoint: string, _options?: RequestInit): Promise => ({})) class ApiRequestError extends Error { constructor(public status: number, message: string) { super(message) this.name = 'ApiRequestError' } } void mock.module('./common.ts', () => ({ apiRequest, ApiRequestError })) function statsFor(good: number, bad: number, pending = 0, unrun = 0) { return { promptVersionId: 'v', status: 'completed', stats: { total: good + bad + pending, good, bad, pending }, reviewedStats: { total: good + bad, good, bad }, testCases: { total: 1 + unrun, completed: 1, pendingReview: 0, partiallyReviewed: 0 }, } } describe('compareVersions', () => { beforeEach(() => apiRequest.mockClear()) test('reports better when the candidate scores higher', async () => { const { compareVersions } = await import('./measure.ts') apiRequest .mockImplementationOnce(async () => statsFor(2, 2)) .mockImplementationOnce(async () => statsFor(4, 0)) const comparison = await compareVersions(['v1', 'v2']) expect(comparison.verdict).toBe('better') expect(comparison.scoreDelta).toBeCloseTo(0.5) expect(comparison.baseline!.score).toBeCloseTo(0.5) expect(comparison.candidate!.score).toBeCloseTo(1) }) test('reports worse when the candidate regresses', async () => { const { compareVersions } = await import('./measure.ts') apiRequest .mockImplementationOnce(async () => statsFor(4, 0)) .mockImplementationOnce(async () => statsFor(1, 3)) expect((await compareVersions(['v1', 'v2'])).verdict).toBe('worse') }) test('reports same for an identical score', async () => { const { compareVersions } = await import('./measure.ts') apiRequest .mockImplementationOnce(async () => statsFor(3, 1)) .mockImplementationOnce(async () => statsFor(6, 2)) const comparison = await compareVersions(['v1', 'v2']) expect(comparison.verdict).toBe('same') expect(comparison.scoreDelta).toBe(0) }) test('is inconclusive when the candidate is only partially reviewed', async () => { const { compareVersions } = await import('./measure.ts') apiRequest .mockImplementationOnce(async () => statsFor(2, 2)) .mockImplementationOnce(async () => statsFor(1, 0, 99)) const comparison = await compareVersions(['v1', 'v2']) // 1 good out of 100 attributes scores 100% on the judged subset; that is not a measurement. expect(comparison.candidate!.score).toBe(1) expect(comparison.verdict).toBe('inconclusive') expect(comparison.scoreDelta).toBeNull() expect(comparison.pending).toBe(99) }) test('is inconclusive when part of the suite never ran', async () => { const { compareVersions } = await import('./measure.ts') apiRequest .mockImplementationOnce(async () => statsFor(4, 0)) .mockImplementationOnce(async () => statsFor(4, 0, 0, 2)) const comparison = await compareVersions(['v1', 'v2']) // The two unrun cases contribute no attributes, so `pending` is 0 and the reviewed // attributes all pass — but the suite is only partly executed. expect(comparison.candidate!.stats.pending).toBe(0) expect(comparison.candidate!.unrunCases).toBe(2) expect(comparison.verdict).toBe('inconclusive') }) test('is inconclusive when the baseline is only partially reviewed', async () => { const { compareVersions } = await import('./measure.ts') apiRequest .mockImplementationOnce(async () => statsFor(1, 0, 50)) .mockImplementationOnce(async () => statsFor(4, 0)) const comparison = await compareVersions(['v1', 'v2']) expect(comparison.verdict).toBe('inconclusive') expect(comparison.baselinePending).toBe(50) }) test('reports coverage per row', async () => { const { compareVersions } = await import('./measure.ts') apiRequest .mockImplementationOnce(async () => statsFor(2, 2)) .mockImplementationOnce(async () => statsFor(1, 0, 99)) const comparison = await compareVersions(['v1', 'v2']) expect(comparison.baseline!.coverage).toBe(1) expect(comparison.candidate!.coverage).toBeCloseTo(0.01) }) test('is inconclusive when the candidate has no judged attributes', async () => { const { compareVersions } = await import('./measure.ts') apiRequest .mockImplementationOnce(async () => statsFor(4, 0)) .mockImplementationOnce(async () => statsFor(0, 0, 4)) const comparison = await compareVersions(['v1', 'v2']) expect(comparison.verdict).toBe('inconclusive') expect(comparison.scoreDelta).toBeNull() expect(comparison.pending).toBe(4) }) test('is inconclusive when the baseline was never reviewed', async () => { const { compareVersions } = await import('./measure.ts') apiRequest .mockImplementationOnce(async () => statsFor(0, 0, 2)) .mockImplementationOnce(async () => statsFor(4, 0)) expect((await compareVersions(['v1', 'v2'])).verdict).toBe('inconclusive') }) test('treats the first id as baseline and the last as candidate', async () => { const { compareVersions } = await import('./measure.ts') apiRequest .mockImplementationOnce(async () => statsFor(1, 3)) .mockImplementationOnce(async () => statsFor(2, 2)) .mockImplementationOnce(async () => statsFor(4, 0)) const comparison = await compareVersions(['v1', 'v2', 'v3']) expect(comparison.rows).toHaveLength(3) expect(comparison.baseline!.promptVersionId).toBe('v1') expect(comparison.candidate!.promptVersionId).toBe('v3') expect(comparison.verdict).toBe('better') }) test('refuses a single version', async () => { const { compareVersions } = await import('./measure.ts') await expect(compareVersions(['v1'])).rejects.toThrow('at least two version ids') }) }) describe('formatVersionComparison', () => { test('renders the table and the delta when both sides are fully reviewed', async () => { const { formatVersionComparison } = await import('./measure.ts') const output = formatVersionComparison({ rows: [ { promptVersionId: 'v1', title: 'v1 vague', stats: { total: 4, good: 2, bad: 2, pending: 0 }, testCases: { total: 1, completed: 1, pendingReview: 0, partiallyReviewed: 0 }, score: 0.5, coverage: 1, unrunCases: 0 }, { promptVersionId: 'v2', title: 'v2 explicit', stats: { total: 4, good: 4, bad: 0, pending: 0 }, testCases: { total: 1, completed: 1, pendingReview: 0, partiallyReviewed: 0 }, score: 1, coverage: 1, unrunCases: 0 }, ], baseline: null, candidate: null, scoreDelta: 0.5, pending: 0, baselinePending: 0, verdict: 'better', }) expect(output).toContain('v1 vague') expect(output).toContain('50%') expect(output).toContain('better (+50 points)') }) test('names which side is unreviewed instead of just saying inconclusive', async () => { const { formatVersionComparison } = await import('./measure.ts') const output = formatVersionComparison({ rows: [{ promptVersionId: 'v1', stats: { total: 100, good: 1, bad: 0, pending: 99 }, testCases: { total: 1, completed: 1, pendingReview: 1, partiallyReviewed: 0 }, score: 1, coverage: 0.01, unrunCases: 0 }], baseline: null, candidate: null, scoreDelta: null, pending: 99, baselinePending: 3, verdict: 'inconclusive', }) expect(output).toContain('inconclusive') expect(output).toContain('3 attribute(s) on the baseline') expect(output).toContain('99 attribute(s) on the candidate') expect(output).toContain('not a measurement') }) test('says so plainly when nothing at all was judged', async () => { const { formatVersionComparison } = await import('./measure.ts') const output = formatVersionComparison({ rows: [{ promptVersionId: 'v1', stats: { total: 0, good: 0, bad: 0, pending: 0 }, testCases: { total: 0, completed: 0, pendingReview: 0, partiallyReviewed: 0 }, score: null, coverage: 0, unrunCases: 0 }], baseline: null, candidate: null, scoreDelta: null, pending: 0, baselinePending: 0, verdict: 'inconclusive', }) expect(output).toContain('neither version has any judged attributes') }) }) describe('getVersionReadiness', () => { beforeEach(() => apiRequest.mockClear()) test('is ready when everything is reviewed and good', async () => { const { getVersionReadiness } = await import('./measure.ts') apiRequest.mockImplementationOnce(async () => statsFor(4, 0)) const readiness = await getVersionReadiness('v1') expect(readiness.ready).toBe(true) expect(readiness.blockers).toEqual([]) expect(readiness.score).toBe(1) }) test('blocks when nothing was reviewed — the batch-before-review case', async () => { const { getVersionReadiness } = await import('./measure.ts') apiRequest.mockImplementationOnce(async () => statsFor(0, 0, 42)) const readiness = await getVersionReadiness('v1') expect(readiness.ready).toBe(false) expect(readiness.blockers).toEqual(['none of the 42 attribute(s) have been reviewed']) }) test('blocks on a partial review and says how much is left', async () => { const { getVersionReadiness } = await import('./measure.ts') apiRequest.mockImplementationOnce(async () => statsFor(1, 0, 41)) const readiness = await getVersionReadiness('v1') expect(readiness.ready).toBe(false) expect(readiness.blockers).toEqual(['41 of 42 attribute(s) are still unreviewed']) }) test('does not confuse "not run yet" with "no structuredOutput"', async () => { const { getVersionReadiness } = await import('./measure.ts') apiRequest.mockImplementationOnce(async () => ({ promptVersionId: 'v', status: 'pending_run', stats: { total: 0, good: 0, bad: 0, pending: 0 }, reviewedStats: { total: 0, good: 0, bad: 0 }, testCases: { total: 3, completed: 0, pendingReview: 3, partiallyReviewed: 0 }, })) const blocker = (await getVersionReadiness('v1')).blockers[0]! expect(blocker).toContain('has not run against it yet') expect(blocker).toContain('no structuredOutput') }) test('blocks when the score is below the bar', async () => { const { getVersionReadiness } = await import('./measure.ts') apiRequest.mockImplementationOnce(async () => statsFor(3, 1)) const readiness = await getVersionReadiness('v1') expect(readiness.ready).toBe(false) expect(readiness.failing).toBe(1) expect(readiness.blockers.join(' ')).toContain('75%') }) test('accepts a deliberately lowered bar', async () => { const { getVersionReadiness } = await import('./measure.ts') apiRequest.mockImplementationOnce(async () => statsFor(3, 1)) expect((await getVersionReadiness('v1', { minScore: 0.7 })).ready).toBe(true) }) test('blocks when there are no test cases at all', async () => { const { getVersionReadiness } = await import('./measure.ts') apiRequest.mockImplementationOnce(async () => ({ promptVersionId: 'v', status: 'pending_run', stats: { total: 0, good: 0, bad: 0, pending: 0 }, reviewedStats: { total: 0, good: 0, bad: 0 }, testCases: { total: 0, completed: 0, pendingReview: 0, partiallyReviewed: 0 }, })) expect((await getVersionReadiness('v1')).blockers.join(' ')).toContain('no test cases') }) test('formats the blockers rather than a bare verdict', async () => { const { formatVersionReadiness } = await import('./measure.ts') const output = formatVersionReadiness({ ready: false, unrunCases: 0, score: null, reviewed: 0, pending: 42, failing: 0, blockers: ['42 attribute(s) are still unreviewed'], }) expect(output).toContain('Not ready to ship') expect(output).toContain('- 42 attribute(s) are still unreviewed') }) }) describe('getVersionReadiness with unrun cases', () => { beforeEach(() => apiRequest.mockClear()) test('blocks when part of the suite produced no result', async () => { const { getVersionReadiness } = await import('./measure.ts') apiRequest.mockImplementationOnce(async () => ({ promptVersionId: 'v', status: 'completed', stats: { total: 4, good: 4, bad: 0, pending: 0 }, reviewedStats: { total: 4, good: 4, bad: 0 }, testCases: { total: 5, completed: 3, pendingReview: 0, partiallyReviewed: 0 }, })) const readiness = await getVersionReadiness('v1') // Every reviewed attribute passed, yet two cases never produced a result. expect(readiness.score).toBe(1) expect(readiness.pending).toBe(0) expect(readiness.unrunCases).toBe(2) expect(readiness.ready).toBe(false) expect(readiness.blockers.join(' ')).toContain('2 of 5 test case(s) produced no result') }) test('counts cases awaiting review as having run', async () => { const { getVersionReadiness } = await import('./measure.ts') apiRequest.mockImplementationOnce(async () => ({ promptVersionId: 'v', status: 'pending_review', stats: { total: 4, good: 0, bad: 0, pending: 4 }, reviewedStats: { total: 0, good: 0, bad: 0 }, testCases: { total: 3, completed: 0, pendingReview: 2, partiallyReviewed: 1 }, })) const readiness = await getVersionReadiness('v1') expect(readiness.unrunCases).toBe(0) expect(readiness.blockers.join(' ')).not.toContain('produced no result') }) })