import type { EventCostMetricsResponse, EventDTO, EventMetricParams, ListEventsParams, ListEventsResponse, UsageClient as UsageClientType, } from '@meistrari/usage' import { getApiBaseUrl, loadFreshApiKey } from './common.ts' // ============================================================================ // Types // ============================================================================ export type UsageCostMetrics = EventCostMetricsResponse export type UsageEvent = EventDTO /** * Filters the usage store accepts. * * Note what is **absent**: there is no `promptVersionId`. Usage is attributed to the canvas, so a * per-version cost cannot be asked for directly — separate versions by time window, or by `model` * when that is what changed. */ export type UsageFilter = Pick< ListEventsParams, 'start' | 'end' | 'projectId' | 'canvasId' | 'promptApplicationId' | 'model' | 'environment' | 'eventType' | 'status' | 'userId' > /** * Roughly how long usage takes to replicate into the query layer after a run finishes. * * Reading cost immediately after running something returns nothing — not because it was free, but * because the event has not landed yet. That empty result is the trap: it reads as "no cost". */ export const USAGE_REPLICATION_LAG_MS = 60_000 export interface UsageWaitResult { events: UsageEvent[] /** * `false` when the wait ended before the expected events arrived — the data may still be * replicating, so an empty or short result says nothing about cost yet. */ complete: boolean waitedMs: number } export interface ModelUsageRow { model: string /** Provider cost. */ cost: number /** What the workspace is billed, after the multiplier. */ effectiveCost: number events: number tokens: number } // ============================================================================ // Client // ============================================================================ let client: UsageClientType | null = null /** * Base URL of the usage service, derived from the Tela API URL unless overridden with * `TELA_USAGE_API_URL`. */ export function getUsageApiBaseUrl(): string { return deriveUsageBaseUrl(getApiBaseUrl(), process.env.TELA_USAGE_API_URL) } /** * The host derivation, separated from the environment so it can be reasoned about directly: * `https://api.telastaging.com` becomes `https://usage-api.telastaging.com`. */ export function deriveUsageBaseUrl(apiBaseUrl: string, override?: string): string { if (override) return override.replace(/\/$/, '') if (apiBaseUrl.includes('localhost')) throw new Error('Set TELA_USAGE_API_URL to reach the usage service from a localhost environment.') return apiBaseUrl.replace(/^https:\/\/api\./, 'https://usage-api.') } async function getClient(): Promise { if (client) return client const { UsageClient } = await import('@meistrari/usage') client = new UsageClient({ baseUrl: getUsageApiBaseUrl(), origin: 'tela-skills' }) return client } // ============================================================================ // Cost // ============================================================================ /** * Authoritative cost for a slice of usage. * * **This is the only trustworthy source of cost.** `creditsUsed` and `usage.cost` on a completion * run are execution-time artifacts — not reconciled, not what the workspace is billed — so they must * never be summed or quoted as spend. * * Always bound the window with `start` and `end`: the store is time-partitioned and an unbounded * query is both slow and prone to failing. */ export async function getUsageCost(filter: UsageFilter = {}): Promise { return (await getClient()).getEventCostMetrics(await loadFreshApiKey(), filter as EventMetricParams) } /** * Individual usage events, each carrying its own cost and token counts. */ export async function listUsageEvents( filter: UsageFilter & { limit?: number, offset?: number } = {}, ): Promise { return (await getClient()).listEvents(await loadFreshApiKey(), filter as ListEventsParams) } /** * Cost grouped by model. * * This is the shape a model comparison needs: the same workload measured per model, from the billing * source rather than from whatever the completion reported. */ export async function getUsageCostByModel( filter: UsageFilter & { limit?: number } = {}, ): Promise { const { data } = await listUsageEvents({ limit: 1000, ...filter }) return groupUsageByModel(data) } /** * Group usage events by model. Separated from the fetch so it can be reasoned about — and tested — * without a network call. */ export function groupUsageByModel(events: Array>): ModelUsageRow[] { const byModel = new Map() for (const event of events) { const model = event.model ?? 'unknown' const row = byModel.get(model) ?? { model, cost: 0, effectiveCost: 0, events: 0, tokens: 0 } row.cost += event.cost ?? 0 row.effectiveCost += event.effectiveCost ?? event.cost ?? 0 row.events += 1 row.tokens += (event.inputAmount ?? 0) + (event.outputAmount ?? 0) byModel.set(model, row) } return [...byModel.values()].sort((a, b) => b.cost - a.cost) } /** * Wait for usage to replicate, then return the events. * * Usage lands in the query layer about a minute after a run finishes, so anything that runs work and * then asks what it cost must wait first. Without this, the answer is a confident zero. * * Returns `complete: false` when the expected events never arrived — report that as "not replicated * yet", never as "no cost". Give it a `start` bounding the window you just executed in. */ export async function waitForUsageEvents( filter: UsageFilter & { limit?: number }, options: { minEvents?: number timeoutMs?: number intervalMs?: number /** Injectable for testing; defaults to {@link listUsageEvents}. */ poll?: (filter: UsageFilter & { limit?: number }) => Promise<{ data: UsageEvent[] }> } = {}, ): Promise { const minEvents = options.minEvents ?? 1 const timeoutMs = options.timeoutMs ?? USAGE_REPLICATION_LAG_MS * 3 const intervalMs = options.intervalMs ?? 10_000 const poll = options.poll ?? listUsageEvents const startedAt = Date.now() let events: UsageEvent[] = [] for (;;) { const { data } = await poll(filter) events = data if (events.length >= minEvents) return { events, complete: true, waitedMs: Date.now() - startedAt } if (Date.now() - startedAt >= timeoutMs) return { events, complete: false, waitedMs: Date.now() - startedAt } await new Promise(resolve => setTimeout(resolve, intervalMs)) } } /** * The workspace's billing position for the current month. */ export async function getCurrentUsage(workspaceId: string): Promise<{ workspaceId: string month: string cost: number credits: number }> { const body = await (await getClient()).getCurrentUsage(await loadFreshApiKey(), workspaceId) return { workspaceId: body.workspace_id, month: body.month, cost: body.cost, credits: body.credits, } } /** * Render a per-model cost table. */ export function formatModelUsage(rows: ModelUsageRow[]): string { if (!rows.length) return 'No usage events matched.' const lines = ['model events tokens cost billed'] for (const row of rows) { lines.push( row.model.slice(0, 28).padEnd(30) + String(row.events).padStart(6) + String(row.tokens).padStart(10) + ` $${row.cost.toFixed(4)}`.padStart(12) + ` $${row.effectiveCost.toFixed(4)}`.padStart(12), ) } return lines.join('\n') }