{"version":3,"file":"extract.d.ts","sourceRoot":"","sources":["../../../src/core/learn/extract.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;GA2BG;AASH,OAAO,KAAK,EAAE,SAAS,EAAgB,MAAM,cAAc,CAAC;AAE5D,OAAO,KAAK,EAAE,aAAa,EAAE,aAAa,EAAiB,MAAM,eAAe,CAAC;AAEjF,OAAO,KAAK,EAAkC,KAAK,EAAE,MAAM,WAAW,CAAC;AACvE,OAAO,KAAK,EAAE,gBAAgB,EAAE,YAAY,EAA4B,gBAAgB,EAAE,MAAM,aAAa,CAAC;AAE9G,OAAO,EAAS,KAAK,UAAU,EAAE,MAAM,YAAY,CAAC;AAEpD,YAAY,EAAE,aAAa,EAAE,aAAa,EAAE,MAAM,eAAe,CAAC;AAClE,OAAO,EAAE,mBAAmB,EAAE,MAAM,WAAW,CAAC;AAChD,YAAY,EAAE,gBAAgB,EAAE,eAAe,EAAE,YAAY,EAAE,gBAAgB,EAAE,MAAM,aAAa,CAAC;AAqBrG;;;;;;;GAOG;AACH,MAAM,WAAW,iBAAiB;IACjC,+CAA+C;IAC/C,IAAI,EAAE,MAAM,EAAE,CAAC;IACf,6CAA6C;IAC7C,WAAW,EAAE,MAAM,EAAE,CAAC;IACtB,4DAA4D;IAC5D,KAAK,EAAE,MAAM,CAAC;IACd,mDAAmD;IACnD,MAAM,EAAE,MAAM,CAAC;IACf,gFAAgF;IAChF,QAAQ,EAAE,MAAM,CAAC;IACjB,8CAA8C;IAC9C,SAAS,EAAE,MAAM,CAAC;IAClB,2DAA2D;IAC3D,UAAU,EAAE,MAAM,CAAC;CACnB;AAED,6FAA6F;AAC7F,MAAM,WAAW,YAAY;IAC5B,uDAAuD;IACvD,MAAM,EAAE,MAAM,CAAC;IACf,2CAA2C;IAC3C,KAAK,EAAE,MAAM,CAAC;IACd,+EAA+E;IAC/E,MAAM,EAAE,MAAM,CAAC;CACf;AAED,MAAM,WAAW,WAAW;IAC3B,eAAe,EAAE,MAAM,CAAC;IACxB,eAAe,EAAE,MAAM,CAAC;IACxB,8DAA8D;IAC9D,IAAI,EAAE,iBAAiB,CAAC;IACxB,gDAAgD;IAChD,MAAM,EAAE,YAAY,CAAC;IACrB;;;;;OAKG;IACH,OAAO,EAAE,OAAO,CAAC;IACjB;;;;;OAKG;IACH,cAAc,EAAE,OAAO,CAAC;IACxB,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,UAAU,EAAE,gBAAgB,EAAE,CAAC;IAC/B,KAAK,EAAE,YAAY,EAAE,CAAC;IACtB,QAAQ,EAAE,gBAAgB,EAAE,CAAC;IAC7B,mFAAmF;IACnF,UAAU,EAAE,MAAM,CAAC;IACnB,iFAAiF;IACjF,GAAG,EAAE,MAAM,CAAC;IACZ;;;;;;;;;;OAUG;IACH,MAAM,EAAE;QACP,mEAAmE;QACnE,UAAU,EAAE,MAAM,CAAC;QACnB,qFAAmF;QACnF,MAAM,EAAE,MAAM,CAAC;QACf,iFAAiF;QACjF,cAAc,EAAE,MAAM,CAAC;KACvB,CAAC;IACF,oEAAoE;IACpE,QAAQ,EAAE,KAAK,CAAC;QAAE,GAAG,EAAE,MAAM,CAAC;QAAC,QAAQ,EAAE,MAAM,CAAC;QAAC,OAAO,EAAE,OAAO,CAAC;QAAC,IAAI,CAAC,EAAE,MAAM,CAAA;KAAE,CAAC,CAAC;CACpF;AAED,MAAM,WAAW,cAAc;IAC9B,GAAG,EAAE,MAAM,CAAC;IACZ,QAAQ,EAAE,MAAM,CAAC;IACjB;;;;OAIG;IACH,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,kFAAkF;IAClF,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,sEAAsE;IACtE,iBAAiB,CAAC,EAAE,MAAM,CAAC;IAC3B,sCAAsC;IACtC,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB;;;OAGG;IACH,KAAK,CAAC,EAAE,UAAU,CAAC;IACnB,kFAAkF;IAClF,WAAW,CAAC,EAAE,OAAO,CAAC;IACtB;;;OAGG;IACH,MAAM,CAAC,EAAE,KAAK,CAAC;QAAE,IAAI,EAAE,MAAM,CAAC;QAAC,WAAW,EAAE,MAAM,CAAA;KAAE,CAAC,CAAC;IACtD,mCAAmC;IACnC,GAAG,CAAC,EAAE,IAAI,CAAC;CACX;AAED,sEAAsE;AACtE,MAAM,WAAW,WAAY,SAAQ,cAAc;IAClD,iDAAiD;IACjD,KAAK,EAAE,KAAK,CAAC;IACb;;;;OAIG;IACH,SAAS,CAAC,EAAE,SAAS,CAAC;IACtB,gFAAgF;IAChF,aAAa,CAAC,EAAE,aAAa,CAAC;IAC9B,mEAAmE;IACnE,UAAU,CAAC,EAAE,CAAC,QAAQ,EAAE;QAAE,IAAI,EAAE,MAAM,CAAC;QAAC,KAAK,EAAE,MAAM,CAAC;QAAC,MAAM,EAAE,MAAM,CAAA;KAAE,KAAK,IAAI,CAAC;IACjF,MAAM,CAAC,EAAE,WAAW,CAAC;CACrB;AA4JD;;;;;;;;;;GAUG;AACH,wBAAgB,oBAAoB,CAAC,OAAO,EAAE,IAAI,CAAC,cAAc,EAAE,KAAK,GAAG,UAAU,GAAG,YAAY,CAAC,GAAG,MAAM,EAAE,CAW/G;AA2HD;;;;GAIG;AACH,wBAAgB,YAAY,CAAC,OAAO,EAAE,cAAc,GAAG,iBAAiB,CAEvE;AAED;;;;;;;;;GASG;AACH,wBAAgB,UAAU,CAAC,OAAO,EAAE,cAAc,GAAG;IAAE,KAAK,EAAE,MAAM,CAAC;IAAC,MAAM,EAAE,MAAM,CAAC;IAAC,OAAO,EAAE,MAAM,CAAA;CAAE,CAUtG;AA4HD,mDAAmD;AACnD,wBAAgB,kBAAkB,CAAC,OAAO,EAAE;IAC3C,GAAG,EAAE,MAAM,CAAC;IACZ,QAAQ,EAAE,MAAM,CAAC;IACjB,MAAM,CAAC,EAAE,KAAK,CAAC;QAAE,IAAI,EAAE,MAAM,CAAC;QAAC,WAAW,EAAE,MAAM,CAAA;KAAE,CAAC,CAAC;CACtD,GAAG,aAAa,CAMhB;AA6HD,0EAA0E;AAC1E,wBAAsB,eAAe,CAAC,OAAO,EAAE,WAAW,GAAG,OAAO,CAAC,WAAW,CAAC,CA0GhF","sourcesContent":["/**\n * Session mining for `/learn`.\n *\n * Reads session `.jsonl` files straight off disk rather than the live context.\n * That is the whole point: the on-disk transcript is complete even when the\n * in-context one has been compacted away, and it spans every past session\n * instead of only this one. Cross-session repetition is the signal that decides\n * whether something is a durable rule or a one-off, and it is the one thing a\n * prompt reading its own context cannot see.\n *\n * This module is the orchestrator, and the split of labour inside it is\n * deliberate:\n *\n * - **Gathering** is deterministic. Finding session files, resolving which cwd\n *   they belong to, walking the active branch of a forked session — all exact,\n *   all cheap, all here.\n * - **Judgement** is the model's, in `mine.ts` and `coverage.ts`. What counts as\n *   a directive, what two phrasings have in common, whether a rule already\n *   covers something — none of that survives contact with a regex, and it used\n *   to be decided by one.\n * - **Counting** is deterministic again, in `reduce.ts`. The number is the\n *   product, and a model asked to count over a long context will be\n *   approximately right.\n *\n * The expensive step is memoized per session file (`cache.ts`), so a session is\n * read by the model exactly once in its life and the counts are still computed\n * over every session in the window on every run.\n */\n\nimport { existsSync, readdirSync, readFileSync, realpathSync, statSync } from \"node:fs\";\nimport { dirname, join, resolve, sep } from \"node:path\";\nimport type { AgentMessage } from \"@kolisachint/hoocode-agent-core\";\nimport { getUserAgentsDir } from \"../../config.js\";\nimport { getSessionDirPath } from \"../session-manager.js\";\nimport { loadSkills } from \"../skills.js\";\nimport { hashSessionFile, pruneLearnCache, readCachedMining, writeCachedMining } from \"./cache.js\";\nimport type { Clusterer, ClusterInput } from \"./cluster.js\";\nimport { fallbackLabel } from \"./cluster.js\";\nimport type { CoverageIndex, CoverageJudge, CoverageQuery } from \"./coverage.js\";\nimport { noCoverageJudge } from \"./coverage.js\";\nimport type { MinableSession, MinedCandidate, Miner } from \"./mine.js\";\nimport type { DirectiveCluster, FixCandidate, MinedSession, Proposable, RequestCandidate } from \"./reduce.js\";\nimport { reduceDirectives, reduceFixes, reduceRequests } from \"./reduce.js\";\nimport { judge, type LearnState } from \"./state.js\";\n\nexport type { CoverageIndex, CoverageMatch } from \"./coverage.js\";\nexport { LEARN_DIGEST_MARKER } from \"./mine.js\";\nexport type { DirectiveCluster, DirectiveStatus, FixCandidate, RequestCandidate } from \"./reduce.js\";\n\n/** Sessions considered, newest first. */\nconst DEFAULT_MAX_SESSIONS = 20;\n/** Sessions older than this are ignored — a pattern that stopped is not a rule. */\nconst DEFAULT_MAX_AGE_DAYS = 30;\n/** Entries parsed per session file, as a guard against pathological transcripts. */\nconst MAX_ENTRIES_PER_SESSION = 8000;\n/** Occurrences a directive needs before it is proposed. */\nconst DEFAULT_MIN_DIRECTIVE_COUNT = 2;\n/**\n * Sessions a request needs before it is worth proposing as a slash command.\n *\n * Higher than the directive bar. A rule you stated twice is a rule; a job you\n * asked for twice may just be a job that came up twice. Three separate sessions\n * is the point at which typing it again is the expensive option.\n */\nconst DEFAULT_MIN_REQUEST_COUNT = 3;\n/** Cap on each list in the digest, so the model's budget goes to the top signals. */\nconst DEFAULT_MAX_PER_CATEGORY = 8;\n\n/**\n * Why a session file on disk did not make it into the digest.\n *\n * \"No recent sessions\" is the one outcome a user cannot act on without this:\n * an empty session directory, a directory full of month-old sessions, and a\n * directory full of sessions belonging to another checkout all produce the same\n * sentence, and the fix differs in each case.\n */\nexport interface SessionScanReport {\n\t/** Directories actually searched, in order. */\n\tdirs: string[];\n\t/** Directories that do not exist on disk. */\n\tmissingDirs: string[];\n\t/** `.jsonl` files found across all searched directories. */\n\tfiles: number;\n\t/** Skipped for being older than the age window. */\n\ttooOld: number;\n\t/** Skipped because the session header records a different working directory. */\n\totherCwd: number;\n\t/** Skipped for being beyond `maxSessions`. */\n\toverLimit: number;\n\t/** Skipped for being unreadable, unparseable, or empty. */\n\tunreadable: number;\n}\n\n/** What the run cost, so the price of an LLM-read pipeline is visible rather than hidden. */\nexport interface MiningReport {\n\t/** Sessions whose candidates came from cache, free. */\n\tcached: number;\n\t/** Sessions sent to the model this run. */\n\tmined: number;\n\t/** Sessions the model failed on. Their signals are missing from the counts. */\n\tfailed: number;\n}\n\nexport interface LearnDigest {\n\tscannedSessions: number;\n\tskippedSessions: number;\n\t/** Where the sessions came from, and what was passed over. */\n\tscan: SessionScanReport;\n\t/** What was read by the model versus reused. */\n\tmining: MiningReport;\n\t/**\n\t * The run stopped before reading the whole window, so the counts below are\n\t * computed from part of it. Callers must not record these as surfaced: a\n\t * partial count can fall under the repeat threshold, and bookmarking it would\n\t * hide the item on the next run, when the evidence is complete.\n\t */\n\taborted: boolean;\n\t/**\n\t * The coverage judge failed, so every directive reads `new` whether or not it\n\t * is written down. Callers must not record these as surfaced either: the\n\t * bookmark stores whether an item was covered when shown, and a wrong `false`\n\t * there tells a later run you passed over a proposal you were never given.\n\t */\n\tcoverageFailed: boolean;\n\toldestSession?: string;\n\tnewestSession?: string;\n\tagentsFilePath?: string;\n\tagentsFileTokens?: number;\n\tdirectives: DirectiveCluster[];\n\tfixes: FixCandidate[];\n\trequests: RequestCandidate[];\n\t/** Items held back because nothing new has happened since they were last shown. */\n\tsuppressed: number;\n\t/** Items that cleared every threshold but lost the ranking to `maxProposals`. */\n\tcut: number;\n\t/**\n\t * What the window contained before the thresholds, so an empty digest can be\n\t * read.\n\t *\n\t * The pipeline filters hard — replayed slash-command bodies, tool output,\n\t * quotes that cannot be found in the transcript, then a distinct-session bar\n\t * — and every one of those is silent. Without these numbers \"nothing to\n\t * propose\" is unreadable: it could mean the sessions taught nothing, or that\n\t * the bar is one session too high, and the reader has no way to tell which\n\t * knob to reach for.\n\t */\n\tfunnel: {\n\t\t/** Occurrences the miner reported and the quote check accepted. */\n\t\tcandidates: number;\n\t\t/** Distinct points after naming — how much the clustering pass actually merged. */\n\t\tpoints: number;\n\t\t/** Points that were named and counted but did not clear the repeat threshold. */\n\t\tbelowThreshold: number;\n\t};\n\t/** Everything this run put on screen, for the caller to persist. */\n\tsurfaced: Array<{ key: string; lastSeen: string; covered: boolean; text?: string }>;\n}\n\nexport interface ExtractOptions {\n\tcwd: string;\n\tagentDir: string;\n\t/**\n\t * An extra directory to scan, normally the live session manager's. The\n\t * per-cwd default directory is always scanned as well, so a session manager\n\t * pointing somewhere unusual cannot hide this directory's history.\n\t */\n\tsessionDir?: string;\n\tmaxSessions?: number;\n\tmaxAgeDays?: number;\n\t/** Occurrences a directive needs before it is proposed. The signal/noise dial. */\n\tminRepeats?: number;\n\t/** Repeats a tool sequence needs before it is proposed as a skill. */\n\tminRequestRepeats?: number;\n\t/** Cap on each list in the digest. */\n\tmaxProposals?: number;\n\t/**\n\t * What previous runs already showed. Items with no new occurrences since are\n\t * held back. Omit (or pass `ignoreState`) to propose everything in the window.\n\t */\n\tstate?: LearnState;\n\t/** Re-propose everything, ignoring what previous runs surfaced (`/learn all`). */\n\tignoreState?: boolean;\n\t/**\n\t * Skills a directive can already be covered by. Defaults to the ones loaded\n\t * from disk; injectable so tests do not read the developer's real skills.\n\t */\n\tskills?: Array<{ name: string; description: string }>;\n\t/** Injectable clock, for tests. */\n\tnow?: Date;\n}\n\n/** Everything the async pipeline needs beyond the window settings. */\nexport interface MineOptions extends ExtractOptions {\n\t/** Reads one session and reports what it saw. */\n\tminer: Miner;\n\t/**\n\t * Names the whole window at once, deciding which occurrences are the same\n\t * point. Without one, each candidate is named after its own wording, which\n\t * groups identical sentences and nothing else.\n\t */\n\tclusterer?: Clusterer;\n\t/** Decides which proposals are already written down. Defaults to \"none are\". */\n\tcoverageJudge?: CoverageJudge;\n\t/** Progress callback, so a cold-cache run is not a silent wait. */\n\tonProgress?: (progress: { done: number; total: number; cached: number }) => void;\n\tsignal?: AbortSignal;\n}\n\ninterface SessionHeaderLike {\n\ttype: \"session\";\n\tid?: string;\n\ttimestamp?: string;\n\tcwd?: string;\n}\n\ninterface EntryLike {\n\ttype: string;\n\tid?: string;\n\tparentId?: string | null;\n\ttimestamp?: string;\n\tmessage?: AgentMessage;\n}\n\n/** One session, reduced to the branch that was actually taken. */\ninterface ParsedSession {\n\tfile: string;\n\tid: string;\n\t/** When the session was opened. Describes the window, not what is in it. */\n\ttimestamp: string;\n\t/**\n\t * When the session was last written to.\n\t *\n\t * This is the clock suppression runs on, and it must not be the session's\n\t * start. A session opened yesterday and worked in today would date everything\n\t * said in it to yesterday, which can be older than the last `/learn` run — so\n\t * something said minutes ago reads as \"nothing new since you were last shown\n\t * this\" and is held back. Per-candidate timestamps would be finer, but the\n\t * miner sees an untimestamped blob and would have to invent them; the\n\t * session's last activity is deterministic, free, and errs toward showing an\n\t * item again rather than hiding it.\n\t */\n\tlastActivity: string;\n\tentries: EntryLike[];\n}\n\n/** Newest entry timestamp on the branch, falling back to when the session opened. */\nfunction lastActivityOf(entries: EntryLike[], fallback: string): string {\n\tlet latest = \"\";\n\tfor (const entry of entries) {\n\t\tif (typeof entry.timestamp === \"string\" && entry.timestamp > latest) latest = entry.timestamp;\n\t}\n\treturn latest || fallback;\n}\n\n/**\n * Reduce a session's raw entries to the branch that was actually taken.\n *\n * Session files are trees — forks and clones append entries that were never\n * part of the same conversation. Walking parent links back from the last entry\n * keeps the miner from reading two turns that never happened in sequence as if\n * they did. Sessions written before entry ids existed are flat, and for those\n * file order *is* the branch.\n */\nfunction activeBranch(entries: EntryLike[]): EntryLike[] {\n\tconst withIds = entries.filter((e) => typeof e.id === \"string\");\n\tif (withIds.length === 0) return entries;\n\n\tconst byId = new Map<string, EntryLike>();\n\tfor (const entry of withIds) byId.set(entry.id as string, entry);\n\n\tconst branch: EntryLike[] = [];\n\tconst seen = new Set<string>();\n\tlet cursor: EntryLike | undefined = withIds[withIds.length - 1];\n\twhile (cursor?.id && !seen.has(cursor.id)) {\n\t\tseen.add(cursor.id);\n\t\tbranch.push(cursor);\n\t\tcursor = cursor.parentId ? byId.get(cursor.parentId) : undefined;\n\t}\n\treturn branch.reverse();\n}\n\n/**\n * Compare two directory paths the way the filesystem does.\n *\n * A session header stores the cwd as it was typed, and the same directory can\n * be spelled several ways: through a symlink (`/tmp` is `/private/tmp` on\n * macOS), with a trailing separator, or in different case on the\n * case-insensitive filesystems that macOS and Windows ship by default. String\n * equality on `resolve()` alone rejects every one of those, and rejecting them\n * here means silently discarding the whole history the command exists to read.\n */\nfunction normalizeDirPath(path: string): string {\n\tlet resolved = resolve(path);\n\ttry {\n\t\tresolved = realpathSync.native(resolved);\n\t} catch {\n\t\t// Deleted or never-created directory: the textual form is all we have.\n\t}\n\t// `resolve` already drops a trailing separator except at a filesystem root,\n\t// where dropping it would turn \"/\" into \"\".\n\tif (resolved.length > 1 && resolved.endsWith(sep)) resolved = resolved.slice(0, -1);\n\treturn process.platform === \"win32\" || process.platform === \"darwin\" ? resolved.toLowerCase() : resolved;\n}\n\nfunction sameDirectory(a: string, b: string): boolean {\n\treturn normalizeDirPath(a) === normalizeDirPath(b);\n}\n\n/** Reason a candidate file produced no session, for the scan report. */\ntype SkipReason = \"otherCwd\" | \"unreadable\";\n\nfunction parseSessionFile(file: string, cwd: string, onSkip: (reason: SkipReason) => void): ParsedSession | undefined {\n\tlet raw: string;\n\ttry {\n\t\traw = readFileSync(file, \"utf-8\");\n\t} catch {\n\t\tonSkip(\"unreadable\");\n\t\treturn undefined;\n\t}\n\n\tconst lines = raw.split(\"\\n\");\n\tlet header: SessionHeaderLike | undefined;\n\tconst entries: EntryLike[] = [];\n\tfor (const line of lines) {\n\t\tif (!line.trim()) continue;\n\t\tif (entries.length >= MAX_ENTRIES_PER_SESSION) break;\n\t\tlet parsed: EntryLike | SessionHeaderLike;\n\t\ttry {\n\t\t\tparsed = JSON.parse(line);\n\t\t} catch {\n\t\t\t// A partially-flushed final line is normal for a live session.\n\t\t\tcontinue;\n\t\t}\n\t\tif (parsed.type === \"session\") {\n\t\t\theader ??= parsed as SessionHeaderLike;\n\t\t\tcontinue;\n\t\t}\n\t\tentries.push(parsed as EntryLike);\n\t}\n\n\t// An explicit `--session` path can put a session for another directory in\n\t// this directory, so trust the header over the file's location.\n\tif (header?.cwd && !sameDirectory(header.cwd, cwd)) {\n\t\tonSkip(\"otherCwd\");\n\t\treturn undefined;\n\t}\n\tif (entries.length === 0) {\n\t\tonSkip(\"unreadable\");\n\t\treturn undefined;\n\t}\n\n\tconst branch = activeBranch(entries);\n\tconst opened = header?.timestamp ?? statSync(file).mtime.toISOString();\n\treturn {\n\t\tfile,\n\t\tid: header?.id ?? file,\n\t\ttimestamp: opened,\n\t\tlastActivity: lastActivityOf(branch, opened),\n\t\tentries: branch,\n\t};\n}\n\n/**\n * Every directory this cwd's sessions could be sitting in.\n *\n * The caller passes the live session manager's directory, which is the right\n * answer almost always — but not quite always, and each exception silently\n * emptied the digest. An in-memory session (`--no-session`) reports `\"\"`; an\n * explicit `--session <path>` reports wherever that file lives; a custom\n * `sessionDir` setting points at one shared directory. In every one of those\n * cases the per-cwd default directory still holds the history worth mining, so\n * search both and let the header check sort out what belongs to this cwd.\n */\nexport function candidateSessionDirs(options: Pick<ExtractOptions, \"cwd\" | \"agentDir\" | \"sessionDir\">): string[] {\n\tconst dirs: string[] = [];\n\tconst seen = new Set<string>();\n\tfor (const dir of [options.sessionDir, getSessionDirPath(options.cwd, options.agentDir)]) {\n\t\tif (!dir) continue;\n\t\tconst key = normalizeDirPath(dir);\n\t\tif (seen.has(key)) continue;\n\t\tseen.add(key);\n\t\tdirs.push(dir);\n\t}\n\treturn dirs;\n}\n\nfunction listSessions(options: ExtractOptions): {\n\tsessions: ParsedSession[];\n\tskipped: number;\n\tscan: SessionScanReport;\n} {\n\tconst dirs = candidateSessionDirs(options);\n\tconst scan: SessionScanReport = {\n\t\tdirs,\n\t\tmissingDirs: [],\n\t\tfiles: 0,\n\t\ttooOld: 0,\n\t\totherCwd: 0,\n\t\toverLimit: 0,\n\t\tunreadable: 0,\n\t};\n\n\tconst maxSessions = options.maxSessions ?? DEFAULT_MAX_SESSIONS;\n\tconst maxAgeDays = options.maxAgeDays ?? DEFAULT_MAX_AGE_DAYS;\n\tconst now = options.now ?? new Date();\n\tconst cutoff = now.getTime() - maxAgeDays * 24 * 60 * 60 * 1000;\n\n\tconst files: string[] = [];\n\tfor (const dir of dirs) {\n\t\tif (!existsSync(dir)) {\n\t\t\tscan.missingDirs.push(dir);\n\t\t\tcontinue;\n\t\t}\n\t\ttry {\n\t\t\tfor (const name of readdirSync(dir)) {\n\t\t\t\tif (name.endsWith(\".jsonl\")) files.push(join(dir, name));\n\t\t\t}\n\t\t} catch {\n\t\t\tscan.missingDirs.push(dir);\n\t\t}\n\t}\n\tscan.files = files.length;\n\n\t// Newest first across all directories, so `maxSessions` keeps the most recent\n\t// history rather than whichever directory happened to be searched first.\n\tconst dated = files\n\t\t.map((file) => {\n\t\t\ttry {\n\t\t\t\treturn { file, mtime: statSync(file).mtime.getTime() };\n\t\t\t} catch {\n\t\t\t\tscan.unreadable++;\n\t\t\t\treturn undefined;\n\t\t\t}\n\t\t})\n\t\t.filter((f): f is { file: string; mtime: number } => !!f)\n\t\t.sort((a, b) => b.mtime - a.mtime);\n\n\tconst sessions: ParsedSession[] = [];\n\tconst seenIds = new Set<string>();\n\tlet skipped = 0;\n\tfor (const { file, mtime } of dated) {\n\t\tif (sessions.length >= maxSessions) {\n\t\t\tscan.overLimit++;\n\t\t\tskipped++;\n\t\t\tcontinue;\n\t\t}\n\t\tif (mtime < cutoff) {\n\t\t\tscan.tooOld++;\n\t\t\tskipped++;\n\t\t\tcontinue;\n\t\t}\n\t\tconst parsed = parseSessionFile(file, options.cwd, (reason) => {\n\t\t\tscan[reason]++;\n\t\t});\n\t\tif (!parsed) {\n\t\t\tskipped++;\n\t\t\tcontinue;\n\t\t}\n\t\t// Searching two directories can turn up the same session twice (an explicit\n\t\t// `--session` path inside the default directory). Counting it twice would\n\t\t// inflate the cross-session repetition that decides what gets proposed.\n\t\tif (seenIds.has(parsed.id)) {\n\t\t\tskipped++;\n\t\t\tcontinue;\n\t\t}\n\t\tseenIds.add(parsed.id);\n\t\tsessions.push(parsed);\n\t}\n\treturn { sessions, skipped, scan };\n}\n\n/**\n * Hold back items already shown that have not recurred since, then cap the rest.\n *\n * Order matters: suppression runs *before* the cap, or an item you already\n * decided on would occupy one of the few slots the digest has and push a live\n * signal off the list.\n */\nfunction applySuppression<T extends Proposable>(\n\titems: T[],\n\tstate: LearnState | undefined,\n\tmaxProposals: number,\n\tcovered: (item: T) => boolean,\n\tonDeclined?: (item: T) => void,\n): { kept: T[]; suppressed: number; cut: number } {\n\tif (!state) {\n\t\treturn { kept: items.slice(0, maxProposals), suppressed: 0, cut: Math.max(0, items.length - maxProposals) };\n\t}\n\n\tconst kept: T[] = [];\n\tlet suppressed = 0;\n\tfor (const item of items) {\n\t\tconst verdict = judge(state, { key: item.key, lastSeen: item.lastSeen, covered: covered(item) });\n\t\tif (verdict.suppressed) {\n\t\t\tsuppressed++;\n\t\t\tcontinue;\n\t\t}\n\t\tif (verdict.previouslyDeclined) onDeclined?.(item);\n\t\tkept.push(item);\n\t}\n\t// Anything past the cap cleared every bar and lost on rank alone. It is not\n\t// suppressed and it is not bookmarked, so it will be back next run — but a\n\t// digest that silently shows eight of twenty reads as \"twenty is all there\n\t// was\", and the reader tunes the wrong knob.\n\treturn { kept: kept.slice(0, maxProposals), suppressed, cut: Math.max(0, kept.length - maxProposals) };\n}\n\n/**\n * Where this cwd's sessions were found and what was passed over, without\n * mining anything. `/learn settings` and `/learn stats` report on the window\n * without paying for a model call.\n */\nexport function scanSessions(options: ExtractOptions): SessionScanReport {\n\treturn listSessions(options).scan;\n}\n\n/**\n * What a run would read, without reading it.\n *\n * Runs the real selection — the same age, cwd, cap and de-duplication rules\n * `mineLearnDigest` applies — and then asks the cache about each survivor. It\n * has to be the same selection: this number is what the confirmation prompt\n * quotes, and a prompt that says twelve before reading three is worse than no\n * prompt at all. Hashing the chosen files is cheap next to sending them to a\n * model.\n */\nexport function planMining(options: ExtractOptions): { total: number; cached: number; pending: number } {\n\tconst { sessions } = listSessions(options);\n\tlet cached = 0;\n\tlet pending = 0;\n\tfor (const session of sessions) {\n\t\tconst hash = hashSessionFile(session.file);\n\t\tif (hash && readCachedMining(options.agentDir, hash)) cached++;\n\t\telse pending++;\n\t}\n\treturn { total: sessions.length, cached, pending };\n}\n\n/** One session's candidates before the naming pass has seen them. */\ninterface RawMinedSession {\n\tsessionId: string;\n\tlastActivity: string;\n\tcandidates: MinedCandidate[];\n}\n\n/**\n * Name every candidate in the window in one place.\n *\n * The pass runs over the whole window at once rather than per session, which is\n * the entire point: \"is this the same point as that\" is unanswerable from\n * inside one transcript. Labels already on record are offered as vocabulary so\n * an item you decided on keeps the key it was bookmarked under — without that,\n * a renamed cluster reads as brand new and suppression quietly stops working.\n *\n * A failed call falls back to naming each candidate after its own wording,\n * which groups identical sentences and nothing else. That is the behaviour the\n * pipeline had before this stage existed, so a clustering outage costs recall,\n * not the run.\n */\nasync function labelSessions(\n\traw: RawMinedSession[],\n\tclusterer: Clusterer | undefined,\n\tknownLabels: string[],\n\tsignal?: AbortSignal,\n): Promise<MinedSession[]> {\n\tconst inputs: ClusterInput[] = [];\n\tconst origin: Array<{ session: number; candidate: number }> = [];\n\tfor (const [sessionIndex, session] of raw.entries()) {\n\t\tfor (const [candidateIndex, candidate] of session.candidates.entries()) {\n\t\t\tinputs.push({ id: inputs.length, kind: candidate.kind, text: candidate.text });\n\t\t\torigin.push({ session: sessionIndex, candidate: candidateIndex });\n\t\t}\n\t}\n\n\tlet labels = new Map<number, string>();\n\tif (clusterer && inputs.length > 0) {\n\t\ttry {\n\t\t\tlabels = await clusterer(inputs, knownLabels, signal);\n\t\t} catch {\n\t\t\t// Fall through to per-text labels below.\n\t\t}\n\t}\n\n\tconst labelled: MinedSession[] = raw.map((session) => ({\n\t\tsessionId: session.sessionId,\n\t\tlastActivity: session.lastActivity,\n\t\tcandidates: [],\n\t}));\n\tfor (const [index, input] of inputs.entries()) {\n\t\tconst where = origin[index];\n\t\tconst candidate = where ? raw[where.session]?.candidates[where.candidate] : undefined;\n\t\tif (!where || !candidate) continue;\n\t\tlabelled[where.session]?.candidates.push({\n\t\t\t...candidate,\n\t\t\tlabel: labels.get(input.id) ?? fallbackLabel(input.text),\n\t\t});\n\t}\n\treturn labelled;\n}\n\n/** Labels already on record, so the naming pass can reuse rather than reinvent them. */\nfunction knownLabelsFrom(state: LearnState | undefined): string[] {\n\tif (!state) return [];\n\treturn Object.keys(state.surfaced).map((key) => key.slice(key.indexOf(\":\") + 1));\n}\n\n/** Nearest AGENTS.md walking up from cwd, so proposals can be checked against it. */\nfunction findAgentsFile(cwd: string): string | undefined {\n\tlet dir = resolve(cwd);\n\twhile (true) {\n\t\tfor (const name of [\"AGENTS.md\", \"AGENTS.MD\", \"CLAUDE.md\", \"CLAUDE.MD\"]) {\n\t\t\tconst candidate = join(dir, name);\n\t\t\tif (existsSync(candidate)) return candidate;\n\t\t}\n\t\tconst parent = dirname(dir);\n\t\tif (parent === dir) return undefined;\n\t\tdir = parent;\n\t}\n}\n\n/**\n * Turn one context file into rule lines the coverage judge can reason about.\n *\n * Headings were dropped and the lines under them sent bare, which asks the\n * model to decide whether a proposal is in scope using text with the scope\n * removed — \"stage only your own files\" reads very differently under \"Git Rules\n * for Parallel Agents\" than on its own. So each line carries its heading path,\n * and the scope it came from, since the corpus spans a repo file and two user\n * ones and a rule's home decides who it binds.\n *\n * Fenced blocks go: a code sample illustrates a rule, it is not one, and on a\n * real file it is a large share of the non-bullet text.\n */\nfunction ruleLinesOf(content: string, scope: string): string[] {\n\tconst lines: string[] = [];\n\tconst headings: string[] = [];\n\tlet inFence = false;\n\n\tfor (const raw of content.split(\"\\n\")) {\n\t\tconst line = raw.trim();\n\t\tif (line.startsWith(\"```\")) {\n\t\t\tinFence = !inFence;\n\t\t\tcontinue;\n\t\t}\n\t\tif (inFence || line.length === 0) continue;\n\n\t\tconst heading = /^(#{1,6})\\s+(.*)$/.exec(line);\n\t\tif (heading) {\n\t\t\tconst depth = heading[1]?.length ?? 1;\n\t\t\theadings.length = Math.min(headings.length, depth - 1);\n\t\t\theadings[depth - 1] = heading[2] ?? \"\";\n\t\t\tcontinue;\n\t\t}\n\n\t\tconst path = headings.filter(Boolean).join(\" > \");\n\t\tlines.push(path ? `[${scope}] ${path} > ${line}` : `[${scope}] ${line}`);\n\t}\n\treturn lines;\n}\n\n/** Assemble the coverage index for a directory. */\nexport function buildCoverageIndex(options: {\n\tcwd: string;\n\tagentDir: string;\n\tskills?: Array<{ name: string; description: string }>;\n}): CoverageIndex {\n\tconst ruleLines: string[] = [];\n\tfor (const file of coverageFiles(options.agentDir, findAgentsFile(options.cwd))) {\n\t\truleLines.push(...ruleLinesOf(file.content, file.scope));\n\t}\n\treturn { ruleLines, skills: options.skills ?? loadSkillIndex(options.cwd, options.agentDir) };\n}\n\n/**\n * Text a proposal is checked against to decide whether it is already written\n * down — the nearest repo context file plus both user scopes.\n *\n * All three matter for suppression, because `/learn` can route a rule to the\n * user scope. Checking only the repo file would report a rule you accepted into\n * `~/.agents/AGENTS.md` as declined.\n */\nfunction coverageFiles(agentDir: string, repoFile: string | undefined): Array<{ scope: string; content: string }> {\n\tconst files: Array<{ scope: string; content: string }> = [];\n\tconst candidates: Array<{ scope: string; path: string | undefined }> = [\n\t\t{ scope: \"repo\", path: repoFile },\n\t\t{ scope: \"user\", path: join(getUserAgentsDir(), \"AGENTS.md\") },\n\t\t{ scope: \"user\", path: join(agentDir, \"AGENTS.md\") },\n\t];\n\tfor (const candidate of candidates) {\n\t\tif (!candidate.path || !existsSync(candidate.path)) continue;\n\t\ttry {\n\t\t\tfiles.push({ scope: candidate.scope, content: readFileSync(candidate.path, \"utf-8\") });\n\t\t} catch {\n\t\t\t// Unreadable context file: treat as absent rather than failing the run.\n\t\t}\n\t}\n\treturn files;\n}\n\n/**\n * Skills a proposal could already have become.\n *\n * `/learn` routes long or conditional guidance to a skill rather than a rule, so\n * without this a proposal you adopted *as a skill* would read as declined —\n * looking only at context files sees an unchanged `AGENTS.md` and concludes you\n * passed. Reuses the real loader rather than a second SKILL.md scanner so the\n * set of locations cannot drift from what the session actually loads.\n */\nfunction loadSkillIndex(cwd: string, agentDir: string): Array<{ name: string; description: string }> {\n\ttry {\n\t\treturn loadSkills({ cwd, agentDir, skillPaths: [], includeDefaults: true }).skills.map((skill) => ({\n\t\t\tname: skill.name,\n\t\t\tdescription: skill.description ?? \"\",\n\t\t}));\n\t} catch {\n\t\t// Skills are an enrichment here, not the point of the command.\n\t\treturn [];\n\t}\n}\n\n/**\n * Run the miner over the window, reusing cached results wherever the file has\n * not changed.\n *\n * A session that fails to mine is counted and skipped rather than aborting the\n * run: one provider hiccup on one transcript should cost that transcript's\n * signals, not the whole digest. The failure count is reported so the reader\n * knows the numbers are short.\n *\n * Cancellation is different from failure and is reported separately. A run\n * stopped half way has counted only some of the window, so its numbers are not\n * merely short — they are wrong in a way that would poison the bookmark if the\n * digest were treated as a completed run.\n */\nasync function mineSessions(\n\tsessions: ParsedSession[],\n\toptions: MineOptions,\n): Promise<{ mined: RawMinedSession[]; report: MiningReport; aborted: boolean }> {\n\tconst mined: RawMinedSession[] = [];\n\tconst report: MiningReport = { cached: 0, mined: 0, failed: 0 };\n\n\tlet done = 0;\n\tfor (const session of sessions) {\n\t\tif (options.signal?.aborted) return { mined, report, aborted: true };\n\n\t\tconst hash = hashSessionFile(session.file);\n\t\tconst cached = hash ? readCachedMining(options.agentDir, hash) : undefined;\n\t\tif (cached) {\n\t\t\tmined.push({ sessionId: session.id, lastActivity: session.lastActivity, candidates: cached.candidates });\n\t\t\treport.cached++;\n\t\t\tdone++;\n\t\t\toptions.onProgress?.({ done, total: sessions.length, cached: report.cached });\n\t\t\tcontinue;\n\t\t}\n\n\t\tconst minable: MinableSession = { id: session.id, timestamp: session.timestamp, entries: session.entries };\n\t\ttry {\n\t\t\tconst candidates = await options.miner(minable, options.signal);\n\t\t\tmined.push({ sessionId: session.id, lastActivity: session.lastActivity, candidates });\n\t\t\treport.mined++;\n\t\t\tif (hash) {\n\t\t\t\twriteCachedMining(options.agentDir, hash, {\n\t\t\t\t\tsessionId: session.id,\n\t\t\t\t\ttimestamp: session.timestamp,\n\t\t\t\t\tcandidates,\n\t\t\t\t\tminedAt: new Date().toISOString(),\n\t\t\t\t});\n\t\t\t}\n\t\t} catch {\n\t\t\t// A cancelled request surfaces here as a rejection. That is not the\n\t\t\t// provider failing on this transcript, so it must not be counted as one.\n\t\t\tif (options.signal?.aborted) return { mined, report, aborted: true };\n\t\t\treport.failed++;\n\t\t}\n\t\tdone++;\n\t\toptions.onProgress?.({ done, total: sessions.length, cached: report.cached });\n\t}\n\n\treturn { mined, report, aborted: false };\n}\n\n/** Apply the coverage verdicts to the clusters they were asked about. */\nfunction applyCoverage(directives: DirectiveCluster[], verdicts: Map<string, { rule?: string; skill?: string }>): void {\n\tfor (const cluster of directives) {\n\t\tconst verdict = verdicts.get(cluster.label);\n\t\tif (!verdict) continue;\n\t\tif (verdict.rule) {\n\t\t\tcluster.status = \"restated\";\n\t\t\tcluster.existingRule = verdict.rule;\n\t\t} else if (verdict.skill) {\n\t\t\tcluster.status = \"has-skill\";\n\t\t\tcluster.existingSkill = verdict.skill;\n\t\t}\n\t}\n}\n\n/** Mine the recent sessions for this cwd and return the ranked digest. */\nexport async function mineLearnDigest(options: MineOptions): Promise<LearnDigest> {\n\tconst { sessions, skipped, scan } = listSessions(options);\n\n\tconst agentsFilePath = findAgentsFile(options.cwd);\n\tlet agentsContent: string | undefined;\n\tif (agentsFilePath) {\n\t\ttry {\n\t\t\tagentsContent = readFileSync(agentsFilePath, \"utf-8\");\n\t\t} catch {\n\t\t\tagentsContent = undefined;\n\t\t}\n\t}\n\n\tconst { mined, report, aborted } = await mineSessions(sessions, options);\n\tpruneLearnCache(options.agentDir, options.now);\n\n\tconst state = options.ignoreState ? undefined : options.state;\n\t// Named against the labels already on record — including in `all` mode, where\n\t// suppression is off but the bookmark still has to line up next run.\n\tconst labelled = await labelSessions(mined, options.clusterer, knownLabelsFrom(options.state), options.signal);\n\n\tconst minRepeats = options.minRepeats ?? DEFAULT_MIN_DIRECTIVE_COUNT;\n\tconst maxProposals = options.maxProposals ?? DEFAULT_MAX_PER_CATEGORY;\n\tconst directives = reduceDirectives(labelled, minRepeats);\n\tconst fixes = reduceFixes(labelled, minRepeats);\n\tconst requests = reduceRequests(labelled, options.minRequestRepeats ?? DEFAULT_MIN_REQUEST_COUNT);\n\n\t// Counted with the threshold at 1, which is the same reduce over the same\n\t// input — so the difference is exactly what the threshold cost, rather than an\n\t// estimate of it.\n\tconst everyPoint =\n\t\treduceDirectives(labelled, 1).length + reduceFixes(labelled, 1).length + reduceRequests(labelled, 1).length;\n\tconst funnel = {\n\t\tcandidates: labelled.reduce((sum, session) => sum + session.candidates.length, 0),\n\t\tpoints: everyPoint,\n\t\tbelowThreshold: everyPoint - directives.length - fixes.length - requests.length,\n\t};\n\n\t// Coverage is asked only about what survived the repeat threshold. Judging\n\t// everything would mean sending the context file alongside a long tail of\n\t// one-off observations that are never going to be proposed.\n\tconst coverage = buildCoverageIndex({ cwd: options.cwd, agentDir: options.agentDir, skills: options.skills });\n\tconst queries: CoverageQuery[] = directives.map((d) => ({ label: d.label, text: d.text }));\n\tlet coverageFailed = false;\n\ttry {\n\t\tconst verdicts = await (options.coverageJudge ?? noCoverageJudge)(queries, coverage, options.signal);\n\t\tapplyCoverage(directives, verdicts);\n\t} catch {\n\t\t// A failed coverage call leaves everything `new`, which over-proposes\n\t\t// slightly. That is the right way to fail: the reader can reject a\n\t\t// duplicate, but cannot recover a proposal that was wrongly withheld. What\n\t\t// must not happen is writing that guess down as if it were a reading.\n\t\tcoverageFailed = true;\n\t}\n\n\tconst timestamps = sessions.map((s) => s.timestamp).sort();\n\t// Directives carry a real coverage signal — is this written down as a rule or\n\t// a skill right now? — which is what separates an adopted proposal from a\n\t// declined one. Fixes and requests do not: a fix may have become a rule, a\n\t// skill, or a habit, and which one is not recoverable here, so they get\n\t// suppression only and are never labelled declined.\n\tconst keptDirectives = applySuppression(\n\t\tdirectives,\n\t\tstate,\n\t\tmaxProposals,\n\t\t(item) => item.status !== \"new\",\n\t\t(item) => {\n\t\t\titem.previouslyDeclined = true;\n\t\t},\n\t);\n\tconst keptFixes = applySuppression(fixes, state, maxProposals, () => false);\n\tconst keptRequests = applySuppression(requests, state, maxProposals, () => false);\n\n\tconst surfaced = [\n\t\t// Directives carry their wording forward so a later `/learn stats` can ask\n\t\t// about coverage using the sentence rather than the slug that names it.\n\t\t...keptDirectives.kept.map((d) => ({\n\t\t\tkey: d.key,\n\t\t\tlastSeen: d.lastSeen,\n\t\t\tcovered: d.status !== \"new\",\n\t\t\ttext: d.text,\n\t\t})),\n\t\t...keptFixes.kept.map((f) => ({ key: f.key, lastSeen: f.lastSeen, covered: false })),\n\t\t...keptRequests.kept.map((r) => ({ key: r.key, lastSeen: r.lastSeen, covered: false })),\n\t];\n\n\treturn {\n\t\tscannedSessions: sessions.length,\n\t\tskippedSessions: skipped,\n\t\tscan,\n\t\tmining: report,\n\t\taborted,\n\t\tcoverageFailed,\n\t\toldestSession: timestamps[0],\n\t\tnewestSession: timestamps[timestamps.length - 1],\n\t\tagentsFilePath,\n\t\tagentsFileTokens:\n\t\t\tagentsContent === undefined ? undefined : Math.round(Buffer.byteLength(agentsContent, \"utf-8\") / 4),\n\t\tdirectives: keptDirectives.kept,\n\t\tfixes: keptFixes.kept,\n\t\trequests: keptRequests.kept,\n\t\tsuppressed: keptDirectives.suppressed + keptFixes.suppressed + keptRequests.suppressed,\n\t\tcut: keptDirectives.cut + keptFixes.cut + keptRequests.cut,\n\t\tfunnel,\n\t\tsurfaced,\n\t};\n}\n"]}