/** * User_Input cleanup detection and correction-effect building. * * The cleanup scan is the human-input counterpart of the System_State repair * scan, and it follows the same system-first rule: SQLite is the authority and * a corrupted User_Input tab (duplicated business keys or anchors, empty-ID * rows, orphan rows) is rewritten from canonical state instead of being * patched row by row. * * Concretely, when the scan proves corruption it builds two kinds of * `user_input` effects on the existing outbox: * * - Every physical row bound in SQLite (row_binding -> active entity -> * canonical user-owned fields) gets a full-row `candidate_reconcile` * rewrite carrying the canonical values with the row's observed cells as * compare-and-set evidence. A binding with a durable active candidate * (sheet_visible_field_state pointer joined to an OPEN/NEEDS_REBASE * conflict) is NEVER planned by this scan: the conflict/candidate evidence * is re-read after the snapshot (and again right before effect building), * and the binding is skipped while a candidate is durable, so a rewrite * can never be enqueued into a conflicted row. The conflict converges * exclusively through resolution, whose binding-keyed reconcile * supersedes any earlier rewrite on the same stream. * - Every physical row NOT matching a bound binding (a duplicate of a bound * key beyond the kept row, empty-ID rows, and orphan rows) gets a * `user_input_delete` carrying the full observed row as its CAS guard. * * Bound-row corrections stream under the BINDING key * (`projection-row::`), the same target key * flush projections and resolution reconciles use, so the resolution's * supersede-and-replan covers any pending cleanup rewrite in the same * transaction and an in-flight flush/reconcile head defers the scan. Only * unbound rows (orphans, empty-ID rows without a binding) keep the physical * anchor as their stream key, because no binding exists to own the stream; * their deletes already write no projection confirmation. * * Duplicated anchors are resolved the way the real provider resolves them: * its preflight anchor index keeps only the FIRST row per anchor value * (indexRows in preflightRows.ts; planEffectBatch mirrors the same rule), so * a scan targets only the first (lowest) row of a duplicate-anchor group for * deletion and defers the surviving rows and the group's rewrite to the next * scan. Each scan therefore converges one surplus row per duplicated anchor, * and a re-scan of a converged tab enqueues nothing (idempotent). * * The scan never writes to the Sheet directly; every correction flows through * the durable outbox and the worker's CAS-guarded slow path. */ import type { NormalizedCell } from "../../../../contracts/encoding/types.js"; import type { NewEffect } from "@hikoutei/ikisaki"; import type { SqlExecutor, SqlStorageAdapter } from "../../../../contracts/storage/sql.js"; import { type SyncSheetsSnapshot } from "../../../../contracts/sheets/syncSheets.js"; /** One snapshot row the cleanup scan may correct. */ export interface CleanupRow { readonly rowNumber: number; /** Physical anchor (row-id system column) value; null when unanchored. */ readonly anchor: string | null; /** Business-key value, or undefined for empty-ID rows. */ readonly identity: string | undefined; /** * Deterministic positive visible revision: the snapshot-provided revision * when present (defensive), otherwise 1. The real provider leaves visible * state to SQLite, so the scan derives a positive baseline revision. */ readonly visibleRevision: number; /** * Visible hash: the snapshot-provided hash when present (defensive), * otherwise computed over the observed cells with computeSyncVisibleHash. */ readonly visibleHash: string; /** Full observed row cells keyed by header (blank cells are null). */ readonly fields: Record; } export type CleanupTargetKind = "duplicate" | "empty_id" | "extra" | "rewrite"; /** One surplus or drifted row proven safe to correct under its CAS guard. */ export interface CleanupTarget { readonly kind: CleanupTargetKind; readonly row: CleanupRow; /** Rewrite-only: canonical SQLite user-owned fields to project. */ readonly canonicalFields?: Record; /** * Rewrite/delete-only: durable row binding id. Rewrites always carry the * binding id so the worker candidate gate can find the binding's visible * state; deletes carry it when a real binding owns the row's anchor; * unbound rows (orphans, empty-ID rows) carry none, so the worker writes * no projection confirmation for a binding that does not exist. */ readonly rowBindingId?: string; } /** Durable evidence consulted before any row may be corrected. */ export interface CleanupEvidence { readonly bindings: readonly CleanupBinding[]; readonly canonical: readonly CleanupCanonicalRow[]; /** row_binding_id -> active candidate hash for user_input visible state. */ readonly candidateHashes: ReadonlyMap; } export interface CleanupBinding { readonly rowBindingId: string; readonly anchorReference: string; readonly state: string; } /** Canonical User_Input projection target owned by one active binding. */ export interface CleanupCanonicalRow { readonly entityId: string; readonly rowBindingId: string; readonly anchorReference: string; readonly fields: Record; readonly fieldRevisionHash: string; } interface CleanupCanonicalSqlShape { readonly entity_id: string; readonly row_binding_id: string; readonly anchor_reference: string; readonly field_name: string; readonly normalized_value: string; } /** * Whole-entity canonical pages are assembled from the storage-owned * entity-batch primitives (same bounded access path as the desired-state * pages: PK-ordered entities, PK-prefix fields, covering binding seeks). * A flat cross-table keyset cannot page here: duplicate `(entity, field)` * keys across several active bindings of one entity would be skipped by a * `(entity_id, field_name)` cursor, silently dropping a binding's fields. */ /** * Stable binding-keyed stream target id shared with flush projections and * resolution reconciles. Bound-row cleanup corrections must stream under the * same key so the resolution's supersede-and-replan lookup and the durable * stream predecessor guard cover them. */ export declare function cleanupBindingStreamTargetId(physicalSheetId: string, rowBindingId: string): string; /** * Reads the set of row bindings carrying a durable active candidate pointer * joined to an OPEN/NEEDS_REBASE conflict. The scan must never plan a * rewrite/delete for one of these bindings: the conflict converges only * through resolution, whose binding-keyed reconcile supersedes any pending * rewrite on the same stream. */ export declare function readCleanupProtectedBindingsWithSql(sql: SqlExecutor, physicalSheetId: string): Promise>; /** Reads the durable binding/canonical/candidate evidence for one tab. */ export declare function readCleanupEvidenceWithSql(sql: SqlExecutor, logicalSheetId: string, physicalSheetId: string): Promise; /** Reads cleanup evidence through a fresh adapter read context. */ export declare function readCleanupEvidence(storage: SqlStorageAdapter, logicalSheetId: string, physicalSheetId: string): Promise; /** Keyset cursor for a paged cleanup binding chunk: the last id already seen. */ export interface CleanupBindingsCursor { readonly rowBindingId: string; } /** Keyset cursor for a paged canonical chunk: the last entity already seen. */ export interface CleanupCanonicalCursor { readonly entityId: string; } /** One entity-batched canonical chunk: whole entities plus progress. */ export interface CleanupCanonicalEntityChunk { readonly rows: readonly CleanupCanonicalSqlShape[]; /** Entities paged (including binding-less ones) — the termination signal. */ readonly entityCount: number; /** Last entity paged — the next cursor (absent only when empty). */ readonly lastEntityId: string | undefined; } /** Reads one bounded page of row bindings in `row_binding_id` order. */ export declare function readCleanupBindingsChunkWithSql(sql: SqlExecutor, logicalSheetId: string, after: CleanupBindingsCursor | undefined, limit?: number): Promise; /** * Reads one bounded chunk of whole entities as flat canonical user-owned * rows, in global entity order. Pass no cursor for the first chunk, then * `{ entityId }` of the last entity paged; an empty chunk (or * `entityCount < limit`) ends the scan. Only `limit` entities are ever * materialized per call. Multi-binding entities emit one row per * (binding, field) with the smallest `row_binding_id` first, so no * binding's fields can be skipped across chunks by construction. */ export declare function readCleanupCanonicalChunkWithSql(sql: SqlExecutor, logicalSheetId: string, after: CleanupCanonicalCursor | undefined, limit?: number): Promise; /** One entity's canonical bindings accumulated across chunk boundaries. */ export interface PartialCleanupCanonical { readonly entityId: string; readonly partials: ReadonlyMap; }>; } /** Completed canonical rows plus the trailing entity still missing fields. */ export interface CleanupCanonicalChunkResult { readonly completed: readonly CleanupCanonicalRow[]; readonly carry: PartialCleanupCanonical | undefined; } /** * Groups one ordered page of canonical rows into completed canonical rows. * * A page may split an entity's fields across the boundary, and several * bindings may share one entity, so the trailing entity is returned as * `carry` (all of its bindings) and must seed the next call; only the flush * after the last page finalizes it. Completed rows hash exactly like the * full-load grouping, in global entity order. */ export declare function assembleCleanupCanonicalChunk(rows: readonly CleanupCanonicalSqlShape[], carry: PartialCleanupCanonical | undefined): CleanupCanonicalChunkResult; /** Finalizes the trailing entity's canonical rows after the last page. */ export declare function flushCleanupCanonicalCarry(carry: PartialCleanupCanonical | undefined): readonly CleanupCanonicalRow[]; /** Snapshot rows grouped by anchor for the chunked cleanup scan. */ export interface CleanupSnapshotIndex { readonly groupsByAnchor: ReadonlyMap; readonly duplicateAnchors: ReadonlySet; } /** * Groups decoded snapshot rows by anchor once per scan. The maps hold * references to the decoded rows; no snapshot data is copied. */ export declare function buildCleanupSnapshotIndex(rows: readonly CleanupRow[]): CleanupSnapshotIndex; /** Streaming evidence for one cleanup scan: small maps plus drifted rewrites. */ export interface CleanupStreamEvidence { readonly bindingsByAnchor: ReadonlyMap; readonly candidateHashes: ReadonlyMap; readonly canonicalAnchors: ReadonlySet; readonly rewriteByAnchor: ReadonlyMap; } /** * Streams binding/canonical/candidate evidence in keyset pages inside the * caller's SQL context. * * Only one page of canonical rows is materialized at a time; completed * canonical rows are rewrite-checked against the snapshot index immediately * and dropped unless drifted. Retained state is small bindings, candidate * hashes, canonical anchor keys, and drifted rewrite targets — never the * full canonical table. */ export declare function readCleanupStreamEvidenceWithSql(sql: SqlExecutor, logicalSheetId: string, physicalSheetId: string, snapshot: CleanupSnapshotIndex): Promise; /** * Classifies snapshot rows into correction targets in full-load order. * * Mirrors `classifyCleanupRows` decision-for-decision (duplicate deletes * first in first-seen anchor order, then snapshot row order for rewrites * and surplus deletes) while reading rewrites from the streamed evidence: * bound non-duplicate anchors resolve through `rewriteByAnchor`, every * other row through the binding map. The target sequence is identical to * the full-load scan, so repair effect ids stay stable across the switch. */ export declare function classifyCleanupStreamTargets(rows: readonly CleanupRow[], snapshot: CleanupSnapshotIndex, evidence: CleanupStreamEvidence): readonly CleanupTarget[]; /** * Decodes snapshot rows into cleanup rows. * * Only anchored rows are eligible: the delete/rewrite effects locate their * CAS target by the physical anchor, so unanchored rows stay on the * observation pipeline instead of being guessed at. The real provider leaves * `visibleRevision`/`visibleHash` ABSENT (visible state lives in SQLite), so * the visible hash is derived from the observed cells with * computeSyncVisibleHash and the revision falls back to a deterministic * positive value; provided values are kept when present (defensive). */ export declare function decodeCleanupRows(snapshot: SyncSheetsSnapshot, identityField: string): readonly CleanupRow[]; /** * Classifies snapshot rows into correction targets with overwrite semantics. * * Decision order: * 1. Duplicated anchor groups: the real provider can only resolve the FIRST * row per anchor (its preflight anchor index keeps the first row only), so * exactly that row is targeted for deletion and every other group member is * deferred to a later scan, when it becomes resolvable. The group's bound * row (if any) is rewritten only after the group is down to one row. * 2. A row whose anchor is bound to an active entity is rewritten from * canonical values unless it already matches (idempotent); an active * candidate hash is attached so the existing candidate guard blocks it. * 3. Rows bound to non-active bindings (candidate without an entity yet, * tombstoned, ambiguous) are protected: there is no canonical value to * rewrite and no proof they are surplus. The one exception is an empty-ID * row under a candidate binding: it can never bind, so it is deleted and * the candidate guard blocks the delete while the candidate is durable. * 4. Every other row (empty-ID rows and orphans, including duplicated orphan * identities and quarantined rows) is surplus relative to SQLite canonical * state and is deleted with its full observed row as CAS evidence; the * durable quarantine/conflict evidence itself is never touched. */ export declare function classifyCleanupRows(rows: readonly CleanupRow[], evidence: CleanupEvidence): readonly CleanupTarget[]; /** Baseline decision for one cleanup target stream. */ export type CleanupBaseline = { readonly kind: "append"; readonly streamSequence: number; readonly supersedeEffectId: string | null; } | { readonly kind: "defer"; }; /** * Resolves the outbox baseline for one cleanup stream. * * Bound rows stream under the BINDING key * (`projection-row::`, the same key flush * projections and resolution reconciles use) so the cleanup correction joins * the row's real predecessor chain and the resolution replan covers it; * unbound rows (orphans, empty-ID rows without a binding) keep the physical * anchor as their stream key. An in-flight head * (pending/processing/delivery_uncertain) defers the scan; a recoverable * failed head stays on the worker retry path; terminal heads (non-recoverable * failed, blocked_candidate, conflict) are superseded with the new correction * in the same transaction so the stream can never wedge. */ export declare function resolveCleanupBaselineWithSql(sql: SqlExecutor, logicalSheetId: string, targetId: string): Promise; /** Effect-building input shared by the cleanup scan orchestrator. */ export interface CleanupEffectContext { readonly storage: SqlStorageAdapter; readonly logicalSheetId: string; readonly physicalSheetId: string; readonly schemaVersion: number; readonly identityField: string; readonly createId: () => string; } /** One built cleanup correction plus the terminal head it supersedes, if any. */ export interface CleanupPlan { readonly effect: NewEffect; readonly supersedeEffectId: string | null; } /** * Builds CAS-carrying corrections for every currently safe target. * * Deletes carry the full observed row (their fields must hash to the observed * visible hash, which the provider's full-row deletion guard requires); * rewrites carry the canonical user-owned fields with the observed row as the * expected CAS state, so the provider only writes when the row is unchanged * since the scan observed it. */ export declare function buildCleanupEffects(context: CleanupEffectContext, sheet: { readonly tabName: string; readonly registeredRange: string; }, targets: readonly CleanupTarget[]): Promise; export {}; //# sourceMappingURL=cleanup.d.ts.map