/** * Document-frequency discriminator-first pattern naming (Layer 2 of the * pattern-name display work) — RENDER-ONLY. * * THE PROBLEM: a 10x pattern name is the `symbolMessage` — a selection of * tokens the engine picked, joined by '_'. On real environments many names * share a long common boilerplate PREFIX (e.g. the OTel-collector resource * envelope) and differ only at the END, so a front-crop renders rows 1/2/6 * identically. A hardcoded OTel denylist (`PREFIX_SKIP`) does not fix it: * that is vendor-specific and silently wrong for Java/MDC, k8s, Datadog, * or arbitrary apps. * * THE GENERAL FIX: boilerplate is whatever is COMMON across the env's * patterns, learned from the data itself — no vendor strip-lists. For each * token compute its document frequency df(tok) = the number of DISTINCT * patterns whose token-set contains it. Tokens with high df are boilerplate * (they appear everywhere, so they don't tell patterns apart); tokens with * low df are discriminators. We surface the discriminators, re-emitted in * the name's ORIGINAL order so tail tokens like `opensearch`/`batch` — the * very ones a front-crop drops — lead the visible name. * * IDENTITY IS UNTOUCHED. Nothing here reads or writes `pattern_hash`, * `templateChars`, `symbol_message`, or any identity field. The output is * purely additive render fields: `display_name` (a cropped string) and * `display_tokens` (the original tokens, each flagged distinctive or not). * * Out of scope (durable follow-up in the ENGINE repo): an engine-stamped * projection that computes this label ONCE after templateChars/templateHash * are frozen, with compile-phase frequency tracking. That is additive and * identity-safe but net-new engine work. This module builds the algorithm * at the MCP, over the live metric label set, now. */ import type { EnvConfig } from './environments.js'; /** Default column budget for a display_name (MCP list / console surface). */ export declare const DEFAULT_NAME_WIDTH = 44; /** * Below this many distinct patterns the df signal is too thin to trust — * one boilerplate run can look as rare as a real discriminator — so we * degrade to Layer 1 (plain mid-ellipsis) rather than collapse on noise. */ export declare const MIN_CORPUS = 20; /** Discriminator tokens surfaced before packing to width. */ export declare const MAX_DISCRIMINATORS = 5; /** * Leading tokens always kept, regardless of document frequency, so a name * says WHAT the statement is and not merely which variant it is. Without * this, a set of patterns sharing a long lead-in loses every shared word to * the boilerplate cutoff and renders as its statistical tail. */ export declare const ANCHOR_TOKENS = 3; /** One render token: the verbatim text + whether df marked it a discriminator. */ export interface DisplayToken { text: string; distinctive: boolean; } /** Render-only naming result. `symbol_message` / `pattern_hash` stay separate + unchanged. */ export interface DisplayName { display_name: string; display_tokens: DisplayToken[]; } /** * Document-frequency context for one environment's pattern set. Built once * and shared across BOTH top_patterns and pattern_detail so the same * pattern renders the SAME display_name on every surface. */ export interface DfContext { /** token -> number of DISTINCT patterns whose token-set contains it. */ dfMap: Map; /** N — number of distinct patterns the df-map was built over. */ patternCount: number; } /** * Split a symbol_message into tokens on '_'. Empty segments (leading, * trailing, or doubled underscores) are dropped. camelCase inside a token * is preserved verbatim — we never Title-case or otherwise mangle * `ValkeyCartStore` / `GET`. */ export declare function tokenizePattern(symbolMessage: string): string[]; /** * Build a df-map from an iterable of symbol_messages. df counts DISTINCT * patterns (a token repeated within one name counts once for that name). */ export declare function buildDfContext(symbolMessages: Iterable): DfContext; /** * Fetch (or reuse) the env's df-context. The pattern set is the distinct * values of the `message_pattern` label active in the window — i.e. every * pattern the env is currently emitting. Cached per `cacheKey` for a short * TTL so both surfaces share one df-map. * * `nowMs` is injectable for tests; production passes Date.now(). * Returns a zero-corpus context (patternCount 0) on any backend failure, so * callers degrade to Layer 1 rather than throw. */ export declare function getEnvDfContext(env: EnvConfig, cacheKey: string, opts?: { windowSeconds?: number; nowMs?: number; }): Promise; /** Test-only: clear the per-env df cache. */ export declare function __clearDfCache(): void; export interface BuildDisplayNameOpts { /** Shared env df-context. Absent/thin corpus => Layer 1 (mid-ellipsis). */ df?: DfContext | null; /** Top service name — its tokens are treated as head/boilerplate. */ service?: string | null; /** Severity (ERROR/INFO/…) — its tokens are treated as head/boilerplate. */ severity?: string | null; /** Column budget (codepoints). Default DEFAULT_NAME_WIDTH (44). */ width?: number; /** Discriminators surfaced before packing. Default MAX_DISCRIMINATORS (5). */ maxDiscriminators?: number; } /** * Compute the discriminator-first display_name + per-token classification * for one symbol_message. RENDER-ONLY. * * Guards (all mandatory): * (a) never blank a name — zero distinctive tokens => mid-ellipsis of the * raw '_'->space name; * (b) length-gate — a name already within the column budget is returned * verbatim (no collapse), which protects short clean names by * construction (`Charge_request_received` is never harmed); * (c) min-corpus floor — no df, or fewer than MIN_CORPUS patterns, * degrades to Layer 1. * Guard (d) (render-time uniqueness across the visible page) is a separate * pass — see `dedupeVisibleNames`. */ export declare function buildDisplayName(symbolMessage: string, opts?: BuildDisplayNameOpts): DisplayName; export interface NameableRow { display_name: string; display_tokens: DisplayToken[]; pattern_hash: string; } /** * After per-row display_names are computed, guarantee they are distinct * across the visible page. For each collision group, append each row's * next-rarest not-yet-shown token (by df asc) until the names diverge; last * resort append ' #'+first4(pattern_hash) — a 4-char suffix, never a * headline. Mutates `display_name` in place. RENDER-ONLY. * * `df` is the shared context used to rank the tie-break tokens; when absent * the tokens are appended in their original order. */ export declare function dedupeVisibleNames(rows: NameableRow[], df?: DfContext | null): void;