/** * POC envelope v2 — input + output sections, no prose. * * The v1 envelope mixed structured data with prose (`headline`, * `human_summary`) and relied on rendered markdown as the canonical * agent-facing surface. The 4-cell experiment showed that giving the * agent prose causes the agent to write a more-confident-sounding but * less-trusted report; the agent's own writing voice is what we want * surfacing the facts. * * v2 returns pure JSON: `input` (proves the moat through measured * scale + methodology) and `output` (rich per-pattern facts the agent * quotes verbatim). No headline. No human_summary. No markdown view. * The agent reads the structured data and writes the report. */ import type { RenderInput } from './poc-report-renderer.js'; import type { IncidentCluster } from './detectors/incident-cluster.js'; import type { PocEnrichment, RedundancyPair } from './poc-enrichers.js'; import type { ExtractedPattern } from './pattern-extraction.js'; import { type Action as CostAction } from './cost.js'; export interface PocEnvelopeV2 { tool: 'log10x_poc_from_siem' | 'log10x_poc_from_local'; schema_version: '2.0'; input: PocInput; output: PocOutput; } export interface PocInput { siem: string; window: { start_iso: string; end_iso: string; duration_seconds: number; }; scope?: string; query?: string; scale: { events_pulled: number; bytes_pulled: number; distinct_patterns_surfaced: number; services_observed: number; pull_wall_time_seconds: number; engine_wall_time_seconds: number; }; stop: { reason: string; /** * Whether pattern discovery saturated (new patterns per 100k events below * 2%). `null` means not measured — nothing computes the discovery curve * yet. It is deliberately nullable rather than defaulting to false: the * previous code derived it from the pull's stop reason, which is a * different question entirely. */ saturation_reached: boolean | null; events_when_saturated: number | null; /** Every sampled sub-window drained. Says nothing about saturation. */ sample_windows_exhausted: boolean; }; coverage: { /** Share of PATTERNS carrying the attribute, 0..1. */ patterns_with_timestamp_pct: number; patterns_with_service_attribution_pct: number; patterns_with_severity_attribution_pct: number; /** Share of EVENTS carrying the attribute, 0..1. */ events_with_timestamp_pct: number; events_with_service_attribution_pct: number; events_with_severity_attribution_pct: number; }; methodology: { engine: 'engine_fingerprint'; fingerprint_determinism: 'stable_across_deploys' | 'sample_lexical'; cost_calculation: 'measured_bytes_x_analyzer_rate'; growth_calculation: 'last_24h_rate_over_window_avg_rate'; incident_clustering: 'jaccard_overlap_pearson_correlation'; first_seen_source: 'per_event_timestamps_in_window' | 'engine_history' | 'unavailable'; }; analyzer_rate_usd_per_gb: number; /** * True when every dollar in this envelope rests on a model rather than on * the destination's own meter. ClickHouse only, today: `analyzer_rate_usd_per_gb` * is a storage rate there and the bill is compute. */ dollars_modeled?: boolean; /** Why the dollars are modeled, in one line. Present only when modeled. */ dollars_modeled_note?: string; } export interface PocOutput { /** * Optional host-agent enrichment block. Populated when the customer * sets `enrich_with_host_agent=true` on the submit and the MCP host * advertises the `sampling` capability. Contributions come from the * host's own LLM using its tools (kubectl, source, dashboards) to * add operational context the engine cannot see. Always * non-throwing: when enrichment is skipped or fails, `metadata` * carries the reason and `contributions` is empty. */ agent_enrichment?: import('./poc-host-agent-enricher.js').AgentEnrichmentResult; aggregates: { totals: { monthly_cost_usd: number; top_n_monthly_cost_usd: number; head_concentration: { top_1: number; top_5: number; top_15: number; }; }; emergence_tally: { new_24h: number; growing: number; stable: number; recent_burst: number; unknown: number; }; by_service: ServiceAggregate[]; by_severity: Record; redundancy_pairs: RedundancyPairOutput[]; }; incidents: IncidentOutput[]; /** * Primary decision surface. One row per service with the consequence * the customer is being asked to commit to: the action, where the bytes * land in plain English, whether they're recoverable, the tool to * recover them, the pattern count rolled up under this service, and * whether the exception list pinned this row to pass. * * The renderer reads this — NOT patterns[] — to build the commitment * artifact body. Patterns surface only in the appendix. */ per_service_consequences: PerServiceConsequence[]; patterns: PatternOutput[]; /** * Feasibility verdict. Populated when the caller passed a * `target_percent_reduction` on the submit. When absent, the POC ran * in recommendation-only mode (no commitment artifact, no verdict). * * `max_achievable_percent` is derived from head_concentration (top-N * share of monthly cost) × per-destination action coverage (which * patterns the level-1 default action can actually reduce) minus the * exception pool (services pinned to action=pass). Feasibility holds * when `max_achievable_percent >= target_percent_reduction`. */ feasibility?: FeasibilityVerdict; /** * Projected commitment artifact. Pre-deploy markdown stub the agent * surfaces alongside the verdict so the buyer sees the contractual * shape of what they would be signing: target, max achievable, * per-action breakdown, exceptions, and the recommended next step * (deploy + configure_engine). */ commitment_artifact?: CommitmentArtifact; /** * Ready-to-commit cap-CSV body in the format `configure_engine` * writes. Composed from per-service aggregates of the per-pattern * recommendations (see `patterns[].actions`). * `configure_engine(from_poc_id=...)` reads this field verbatim * instead of re-deriving the policy from `patterns[].actions`. * Emitted only when the POC ran with a `target_percent_reduction` * AND the feasibility verdict was reached. * * Row grammar (the ENGINE's cap grammar — rate-object-cap.js): * container,cap ← header * ,:: ← one row per service * * Container-keyed only: the engine matches rows against the * k8s_container value on the wire, so `pat:` rows are dead. The * cap is never 0 — a 0 cap is a per-container regulator opt-out. */ cap_csv?: string; /** * Sibling actions.csv body — the ENGINE's per-service action file * (rateReceiverActionLookupFile). Grammar: * container,action ← header * ,:: ← one row per service * The engine reads the over-cap disposition ONLY from this file; a * service with no row defaults to `drop`. Ships with cap_csv on every * delivery. */ actions_csv?: string; } export interface FeasibilityVerdict { feasible: boolean; target_percent_reduction: number; max_achievable_percent: number; /** Plain-English explanation of how max_achievable_percent was derived. */ reason: string; /** Per-action breakdown of the achievable pool (in monthly $). */ achievable_by_action: Array<{ action: CostAction; monthly_cost_usd: number; pattern_count: number; }>; /** Services excluded from the achievable pool (pinned to action=pass). */ exception_services: string[]; /** Monthly cost (USD) covered by the exception list. */ exception_monthly_cost_usd: number; /** * Categories the POC did NOT verify. The agent must surface this * verbatim in the commitment artifact so it can't hallucinate that * "the tool checked everything." Default set below covers the four * highest-risk omissions. */ not_checked: string[]; /** Plain-English version of not_checked, ready to paste into the artifact. */ not_checked_statement: string; } export interface PerServiceConsequence { service: string; action: CostAction; total_bytes_per_month: number; destination_description: string; recoverable: boolean; recover_via: string | null; pattern_count: number; exception_applied: boolean; } export interface CommitmentArtifact { /** Markdown block, pre-deploy framing. */ markdown: string; /** Recommended next-step tool call for the agent to chain into. */ next_step: { tool: 'log10x_advise_install' | 'log10x_configure_engine'; reason: string; }; } export interface ServiceAggregate { service: string; monthly_cost_usd: number; pattern_count: number; top_pattern_index: number; approx_bytes_per_sec: number; share_of_total: number; } export interface RedundancyPairOutput { pattern_a_index: number; pattern_b_index: number; count_ratio: number; min_count: number; service: string; hypothesis: 'same_business_event_logged_twice' | 'http_request_response_pair' | 'enter_exit_pair' | 'unknown_pair'; } export interface IncidentOutput { id: number; service: string; representative_descriptor: string; join_signal: 'jaccard_direct' | 'overlap_shared' | 'jaccard_with_correlation'; confidence: number; member_pattern_indices: number[]; combined_monthly_cost_usd: number; root_cause_hypothesis: string; } export interface PatternOutput { rank: number; identity: string; fingerprint_hash: string; service: string | null; severity: string | null; metrics: { events_in_window: number; events_per_day_avg: number; events_last_24h: number; bytes_in_window: number; cost_per_month_usd: number; cost_per_year_usd: number; share_of_total: number; }; emergence: { category: 'new' | 'growing' | 'stable' | 'recent_burst' | 'unknown'; /** * First occurrence of this pattern WITHIN THE STRATIFIED SAMPLE * the connector pulled, not the pattern's true first emission * time. The connector's 24 random sub-windows cover only ~25% of * the requested window, so values here can lag the customer's * actual first-emission by days. Use this for relative ordering, * not as a definitive "this pattern appeared at time T" claim. */ first_seen_in_sample_iso: string | null; last_seen_in_sample_iso: string | null; age_in_sample_days: number | null; acceleration_ratio: number; duration_in_sample_days: number | null; events_by_hour_sparkline: number[] | null; }; top_slot: { name: string; distinct_count: number; distinct_over_event_count: number; unbounded: boolean; } | null; incident_cluster_id: number | null; redundancy_partner_indices: number[]; actions: PatternActions; } /** * Per-pattern action recommendation, expressed in the 6-action vocab the * cap-CSV writer / parser share. The renderer picks `recommended_action` * by combining `DEFAULT_ACTION_BY_DESTINATION[siem]` with the head- * concentration heuristic — high-volume info-class patterns land on * the destination's level-1 action (Datadog → tier_down, Splunk → * offload, ClickHouse → offload, …); error/audit and exception-pinned * patterns land on `pass`; mid-volume info patterns land on `sample`. * * Replaces the prior bag of sub-action shapes (code_fix / * forwarder_exclusion / siem_exclusion / compact / regulate_cap). The * new shape is a single recommendation per pattern, ready to compose * directly into the cap-CSV `pat:,:::` row. */ export interface PatternActions { /** 6-action recommendation for this pattern on this destination. */ recommended_action: CostAction; /** * Plain-prose explanation of why this action was selected. Cites the * destination level-1 lever, head-concentration band, and any * exception-service / floor pin in effect. */ reason: string; /** * Projected monthly dollar savings if the recommendation is committed, * computed using the same reduction coefficients the feasibility * verdict uses (drop=1.0, offload=1.0, compact=0.7, tier_down=0.6, * sample=0.9, pass=0). Real dollars, rounded to cents. */ expected_savings_usd_per_month: number; /** * Sample-keep denominator. Populated only when * `recommended_action === 'sample'`; null otherwise. Matches the * sampleN argument the cost lib uses (`bytes_out = bytes_in / N`). */ sample_n: number | null; /** * Cap, expressed in bytes per 4-minute reset window, that * configure_engine would write for this pattern. Used to construct * the `pat:,:::` cap-CSV row. The * configure_engine consumer reads `cap_csv` directly; this field is * the per-pattern view for agents inspecting individual rows. */ cap_bytes_per_window: number; /** * Plain-English consequence of this pattern's recommended action. * Mirrors per_service_consequences[].destination_description but at * pattern grain. Renderer hides this from the commitment artifact * body — it surfaces only in the "by pattern" appendix and feeds the * savings report + overflow contents view + audit trail. */ consequence: { destination_description: string; recoverable: boolean; recover_via: string | null; bytes_per_month: number; }; } /** * Build the v2 envelope from RenderInput + the already-enriched * patterns / clusters / redundancy pairs produced by the renderer's * enrichPatternsWithSections helper. Reuses every computation that * already happened — no double work. */ export declare function buildPocEnvelopeV2(input: RenderInput, enrichedPatterns: Array, clusters: IncidentCluster[], redundancyPairs: RedundancyPair[], topN: number, opts?: { /** * Customer-specified reduction target (0-100). When present, the * envelope emits a feasibility verdict + commitment artifact stub. * When absent, the POC stays in recommendation-only mode. */ targetPercentReduction?: number; /** * Services flagged to stay in the SIEM with full retention. Patterns * whose service is in this list are pinned to action=pass and their * bytes are subtracted from the achievable pool. */ exceptionServices?: string[]; /** * Service-level action overrides. Applied AFTER destination default * and AFTER exception_services. Map of service name → action. */ pinServices?: Record; /** * Per-pattern action overrides (advanced). Applied AFTER pinServices. * Map of pattern_hash → action. */ pinPatterns?: Record; }): PocEnvelopeV2;