import { apiRequest } from './common.ts' import type { Generation, TestCase, TestCaseWithGeneration } from './test-cases.ts' // ============================================================================ // Types // ============================================================================ /** 1 = thumbs up, 0 = thumbs down, null = clear the review of this attribute. */ export type AttributeFeedback = 1 | 0 | null export interface AttributeFeedbackEntry { /** * A dot-path into the prompt version's structured output schema: `total`, `address.city`. * Arrays are a single key (`ingredients`), not one key per element. * * Never validated server-side — a typo is stored and then invisible everywhere. */ attributeKey: string feedback: AttributeFeedback } export interface GenerationComment { id: string content: string generationId: string promptVersionId: string createdBy: string createdAt: string updatedAt: string deletedAt: string | null } export interface ValidationRule { id: string description: string createdAt: string createdBy?: string } export interface TestCaseTag { id: string name: string color: string promptId: string | null agentId: string | null workspaceId: string createdAt: string updatedAt: string deletedAt: string | null } export type TestCaseReviewStatus = | 'pending_run' | 'running' | 'pending_review' | 'partially_reviewed' | 'completed' | 'aborted' | 'timeout' | 'failed' export interface TestCaseReview { status: TestCaseReviewStatus /** Attribute keys still awaiting a verdict. */ pending: string[] reviewed: string[] /** True when the prompt version declares no reviewable attributes at all. */ unreviewable: boolean } // ============================================================================ // Review — per-attribute feedback // ============================================================================ /** * Record a thumbs verdict on one output attribute of a generation. * * This is what makes a result *reviewed* rather than merely executed: it writes * `generation.attributeFeedback`, appends to `generation.metadata.reviewers`, and feeds the test * case's answer bank (`testCase.answers[key].good/bad`), which auto-classifies future runs. * * Do not write `attributeFeedback` through `PATCH /generation/:id` — that path skips all three. * * Only call this on a finished generation whose content is JSON. The endpoint parses the content * unguarded and returns 500 for an empty or non-JSON body. */ export async function updateAttributeFeedback( generationId: string, attributeKey: string, feedback: AttributeFeedback, ): Promise<{ generation: Generation, testCase: TestCase }> { return apiRequest(`/generation/${generationId}/attribute-feedback`, { method: 'PATCH', body: JSON.stringify({ attributeKey, feedback }), }) } /** * Record verdicts on several attributes in one call (1..1000). * * No-op entries are skipped silently, so re-submitting an unchanged verdict will not refresh the * reviewer timestamp. */ export async function updateAttributeFeedbackBulk( generationId: string, feedbacks: AttributeFeedbackEntry[], ): Promise<{ generation: Generation, testCase: TestCase }> { return apiRequest(`/generation/${generationId}/attribute-feedbacks-bulk`, { method: 'PATCH', body: JSON.stringify({ feedbacks }), }) } // ============================================================================ // Review — comments // ============================================================================ /** * Leave a free-text comment on a generation. Distinct from thumbs feedback: a comment carries the * explanation, a thumbs-down does not, and comments take no part in the review status. * * `createdBy` is filled server-side from the caller. */ export async function createGenerationComment(payload: { content: string generationId: string promptVersionId: string }): Promise { return apiRequest('/generation-feedback', { method: 'POST', body: JSON.stringify(payload), }) } export async function deleteGenerationComment(commentId: string): Promise { await apiRequest(`/generation-feedback/${commentId}`, { method: 'DELETE' }) } /** * Read the comments on a test case's generations. * * `GET /generation-feedback` cannot be used: it requires a `createdBy` filter and its controller * filters on `deletedAt IS NOT NULL`, so it only ever returns deleted comments. The live comments * come back attached to the generation on `GET /test-case/:id` instead. */ export async function listGenerationComments(testCaseId: string): Promise { const testCase = await apiRequest(`/test-case/${testCaseId}`) return (testCase.generations ?? []) .flatMap((generation) => { const feedback = (generation as Generation & { userFeedback?: GenerationComment | GenerationComment[] }).userFeedback if (!feedback) return [] return Array.isArray(feedback) ? feedback : [feedback] }) } // ============================================================================ // Review — validation rules // ============================================================================ /** * Attach an acceptance criterion to one output attribute. * * Rules are not just documentation: the test case validator runs them through an LLM judge and * writes the verdict into `generation.validations[attributeKey]`, which counts as a review. * Write them as checkable statements. * * **Never call this concurrently for the same test case.** The endpoint reads the test case's whole * `validationRules` object, merges one rule in, and writes it back, with no locking — so parallel * calls silently overwrite each other and most of the rules vanish. Use {@link addValidationRules}, * or await each call. */ export async function addValidationRule( testCaseId: string, attributeKey: string, description: string, ): Promise<{ rule: ValidationRule, testCase: TestCase }> { return apiRequest(`/test-case/${testCaseId}/validation-rules/${encodeURIComponent(attributeKey)}`, { method: 'POST', body: JSON.stringify({ description }), }) } /** * Attach several criteria to one test case, one at a time. * * Sequential by construction: the endpoint's read-modify-write on the test case's rule object loses * writes under concurrency. Reads the test case back afterwards and throws when the stored rules do * not match what was sent, because a partial write returns `200` and looks like success. */ export async function addValidationRules( testCaseId: string, rules: Array<{ attributeKey: string, description: string }>, ): Promise { const created: ValidationRule[] = [] for (const { attributeKey, description } of rules) { const { rule } = await addValidationRule(testCaseId, attributeKey, description) created.push(rule) } const testCase = await apiRequest }>( `/test-case/${testCaseId}`, ) const stored = Object.values(testCase.validationRules ?? {}) .reduce((total, entry) => total + (entry.rules?.length ?? 0), 0) if (stored < rules.length) { throw new Error( `Only ${stored} of ${rules.length} validation rules persisted on ${testCaseId}. ` + 'The endpoint overwrites concurrent writes; re-apply the missing ones before trusting any score.', ) } return created } /** * Remove a rule. Deleting one that does not exist still returns 200. */ export async function deleteValidationRule( testCaseId: string, attributeKey: string, ruleId: string, ): Promise<{ testCase: TestCase }> { return apiRequest(`/test-case/${testCaseId}/validation-rules/${encodeURIComponent(attributeKey)}/${ruleId}`, { method: 'DELETE', }) } // ============================================================================ // Tags // ============================================================================ /** * Tags for a canvas or workflow's test cases. A tag belongs to exactly one prompt — there are no * workspace-wide tags, and agent tags live on a separate surface. */ export async function listTestCaseTags(promptId: string): Promise { return apiRequest(`/test-case-tags?promptId=${promptId}`) } /** * `color` is required and unvalidated (any string). Names are not unique. */ export async function createTestCaseTag(payload: { name: string color: string promptId: string }): Promise { return apiRequest('/test-case-tags', { method: 'POST', body: JSON.stringify(payload), }) } export async function updateTestCaseTag( tagId: string, payload: Partial<{ name: string, color: string }>, ): Promise { return apiRequest(`/test-case-tags/${tagId}`, { method: 'PATCH', body: JSON.stringify(payload), }) } export async function deleteTestCaseTag(tagId: string): Promise { await apiRequest(`/test-case-tags/${tagId}`, { method: 'DELETE' }) } export async function getTestCaseTags(testCaseId: string): Promise { return apiRequest(`/test-case/${testCaseId}/tags`) } /** Returns no body — re-read `getTestCaseTags` to confirm. */ export async function addTestCaseTags(testCaseId: string, tagIds: string[]): Promise { await apiRequest(`/test-case/${testCaseId}/tags`, { method: 'POST', body: JSON.stringify({ tagIds }), }) } export async function removeTestCaseTag(testCaseId: string, tagId: string): Promise { await apiRequest(`/test-case/${testCaseId}/tags/${tagId}`, { method: 'DELETE' }) } /** Test case ids outside the caller's workspace are skipped silently rather than rejected. */ export async function bulkAddTestCaseTags(testCaseIds: string[], tagIds: string[]): Promise { await apiRequest('/test-case/bulk-add-tags', { method: 'POST', body: JSON.stringify({ testCaseIds, tagIds }), }) } export async function bulkRemoveTestCaseTags(testCaseIds: string[], tagIds: string[]): Promise { await apiRequest('/test-case/bulk-remove-tags', { method: 'POST', body: JSON.stringify({ testCaseIds, tagIds }), }) } // ============================================================================ // Local status derivation // ============================================================================ export interface SchemaProperty { type?: string properties?: Record hideFromValidation?: boolean } /** * Flatten a structured output schema to the dot-paths used as attribute keys. * Arrays stay a single key; only leaves are emitted. */ function flattenSchema(properties: Record, prefix = ''): string[] { return Object.entries(properties).flatMap(([name, property]) => { if (property?.hideFromValidation) return [] const key = prefix ? `${prefix}.${name}` : name return property?.type === 'object' && property.properties ? flattenSchema(property.properties, key) : [key] }) } function flattenContent(content: unknown, prefix = ''): string[] { if (content === null || typeof content !== 'object' || Array.isArray(content)) return prefix ? [prefix] : [] return Object.entries(content as Record).flatMap(([name, value]) => { const key = prefix ? `${prefix}.${name}` : name return value !== null && typeof value === 'object' && !Array.isArray(value) ? flattenContent(value, key) : [key] }) } /** * Work out whether a test case result has actually been reviewed, and what is still outstanding. * * The product has no `reviewed` status: a fully reviewed result and one with nothing to review both * read as `completed`. This returns `unreviewable: true` for the second case so the two can be told * apart — the distinction behind test cases that look validated but never were. * * An attribute counts as reviewed when it carries manual `attributeFeedback` (0 or 1) or an * inferred `validations[key].type` of `good` or `bad`. `warning` and `missing` do not count. * * Pass the prompt version the generation ran against — `getCanvasVersion(generation.promptVersionId)`. * Reviewable attributes come from its `configuration.structuredOutput.schema.properties`; a * workflow with an empty schema falls back to the fields present in the generation content. */ /** * The most recent generation of a test case. * * `testCase.generations` is **not** in chronological order — re-running leaves the superseded * generation in the array, sometimes last. Feeding `generations.at(-1)` to a review call writes * feedback onto an aborted run that nobody will ever look at. */ export function getLatestGeneration(testCase: TestCaseWithGeneration): Generation | null { return [...(testCase.generations ?? [])] .sort((a, b) => new Date(b.createdAt).getTime() - new Date(a.createdAt).getTime())[0] ?? null } export interface ReviewablePromptVersion { isWorkflow?: boolean | null configuration?: { structuredOutput?: { enabled?: boolean | string schema?: { properties?: Record } } } | null } export function getTestCaseReview( testCase: TestCaseWithGeneration, promptVersion: { isWorkflow?: boolean | null configuration?: { structuredOutput?: { enabled?: boolean | string, schema?: { properties?: Record } } } | null } | null | undefined, ): TestCaseReview { const generation = getLatestGeneration(testCase) if (!generation) return { status: 'pending_run', pending: [], reviewed: [], unreviewable: true } if (generation.status === 'aborted' || generation.status === 'timeout' || generation.status === 'failed') return { status: generation.status, pending: [], reviewed: [], unreviewable: true } if (generation.status === 'running' || generation.status === 'finalizing') return { status: 'running', pending: [], reviewed: [], unreviewable: false } if (!generation.content) return { status: 'pending_run', pending: [], reviewed: [], unreviewable: false } const structuredOutput = promptVersion?.configuration?.structuredOutput // `enabled` is a BooleanString: the persisted value may be the string 'false'. const enabled = structuredOutput?.enabled === true || structuredOutput?.enabled === 'true' if (!enabled) return { status: 'completed', pending: [], reviewed: [], unreviewable: true } let fields = flattenSchema(structuredOutput?.schema?.properties ?? {}) if (!fields.length && promptVersion?.isWorkflow) { try { fields = flattenContent(JSON.parse(generation.content)) } catch { fields = [] } } if (!fields.length) return { status: 'completed', pending: [], reviewed: [], unreviewable: true } const isReviewed = (key: string) => { const feedback = generation.attributeFeedback?.[key] const validation = generation.validations?.[key]?.type return feedback !== undefined || validation === 'good' || validation === 'bad' } const reviewed = fields.filter(isReviewed) const pending = fields.filter(key => !isReviewed(key)) if (!pending.length) return { status: 'completed', pending, reviewed, unreviewable: false } return { status: reviewed.length ? 'partially_reviewed' : 'pending_review', pending, reviewed, unreviewable: false, } }