/** * jev-tool-ranker — ranks tools for `tool_search` and codemode's `searchTools()` with Jev. * * Each candidate tool is one yes/no question ("would this tool help with the request?"), batched * into parallel requests; the model's probability is the score. This is the `toolSearch` route of * `jevAdvisory` (on by default). It asks the shared Jev model; set * `jevAdvisory.routes.toolSearch.model` to use another decisions model from `jev/classifier.ts`. * The ranker never throws and never blocks discovery: with no * session, the route off, no Token-In credential, or any request failing, it returns the local * BM25 ranking, and tools the model did not score are appended in BM25 order. * * Which tools are sent: every candidate, up to TOOL_RANK_MAX_REQUESTS * FIND_BATCH. Past that cap * the BM25 order decides which candidates are sent; the rest keep their BM25 place. */ import { Bm25Ranker, getSettingsPath, registerToolRanker } from "@selesai/code"; import type { ExtensionAPI, ToolRankOptions, ToolRanker, ToolSearchDocument, ToolSearchMatch, } from "@selesai/code"; import { FIND_BATCH, FindRequests, type JudgeItem, type JudgeStage, judgeNouls } from "./jev-ask-tool.ts"; import { askJevAnswers, confidenceBucket, emitJevTelemetry, type JevAdvisoryConfig, jevConnection, type JevRuntime, jevUnavailable, readJevAdvisoryConfig, } from "./jev/decisions.ts"; /** Parallel Jev requests one ranking may send (16 tools each). */ export const TOOL_RANK_MAX_REQUESTS = 12; /** Spare requests for retrying transport failures. */ export const TOOL_RANK_RETRIES = 2; /** A tool is a match at this probability or above: matches become part of the model's context. */ export const TOOL_RANK_KEEP = 0.5; /** Characters of a tool's summary shown to Jev. */ export const TOOL_RANK_SUMMARY_CHARS = 480; const STAGE: Pick = { stateKey: "tools", criteria: { true: "The tool's name and description show it can do what the request asks for, or is a direct step toward it.", false: "The tool does something unrelated to the request.", }, }; export interface JevToolRankerOptions { /** Receives route telemetry (shape and outcome only; never the query or tool text). */ events?: { emit(channel: string, data: unknown): void }; /** Ranks when Jev cannot, and orders the candidates sent past the cap. Default: BM25. */ fallback?: ToolRanker; /** Reads the `jevAdvisory` settings. Default: the agent settings file. */ readConfig?: () => JevAdvisoryConfig; } /** Resolves with `work`, or rejects if `signal` aborts first. The request itself is not cancelled. */ function abortable(work: Promise, signal: AbortSignal | undefined): Promise { if (!signal) return work; return new Promise((resolve, reject) => { const onAbort = () => reject(new Error("aborted")); if (signal.aborted) return onAbort(); signal.addEventListener("abort", onAbort, { once: true }); work.then(resolve, reject).finally(() => signal.removeEventListener("abort", onAbort)); }); } export function createJevToolRanker(options: JevToolRankerOptions = {}): ToolRanker { const fallback = options.fallback ?? new Bm25Ranker(); const readConfig = options.readConfig ?? (() => readJevAdvisoryConfig(getSettingsPath())); return { async rank( query: string, documents: readonly ToolSearchDocument[], limit: number, rankOptions?: ToolRankOptions, ): Promise { const local = await fallback.rank(query, documents, documents.length, rankOptions); const ctx = rankOptions?.ctx; if (!ctx || documents.length === 0 || limit <= 0 || query.trim() === "") return local.slice(0, limit); const started = Date.now(); const telemetry = (outcome: "jev" | "fallback", extra: Record = {}) => emitJevTelemetry(options.events, "decision", { route: "toolSearch", outcome, candidates: documents.length, elapsedMs: Date.now() - started, ...extra, }); try { // Same cast as ask_jev: the session registry is wider than JevRuntime's header type. const runtime = ctx as unknown as JevRuntime; const config = readConfig(); const route = config.routes.toolSearch; if (!route.enabled) return local.slice(0, limit); const connection = jevConnection(config, route); const unreachable = await jevUnavailable(runtime, connection); if (unreachable) { telemetry("fallback", { confidence: confidenceBucket(undefined), reason: unreachable }); return local.slice(0, limit); } // BM25 hits first, then the rest in document order: the cap drops the least likely tools. const byName = new Map(documents.map((document) => [document.name, document])); const hits = new Set(local.map((match) => match.name)); const order = [...local.map((match) => match.name), ...documents.map((d) => d.name).filter((n) => !hits.has(n))]; const judged = order.slice(0, TOOL_RANK_MAX_REQUESTS * FIND_BATCH).map((name) => byName.get(name)!); const items: JudgeItem[] = judged.map((document) => ({ key: document.name, text: (document.summary ?? document.text).slice(0, TOOL_RANK_SUMMARY_CHARS), question: `Would calling the tool "${document.name}" help carry out the request?`, })); const requests = new FindRequests( (payload) => askJevAnswers(runtime, connection, { payload, maxBytes: route.payloadBytes }), TOOL_RANK_MAX_REQUESTS + TOOL_RANK_RETRIES, ); const outcome = await abortable( judgeNouls(requests, { request: query, ...STAGE }, items, route.payloadBytes, { send: TOOL_RANK_MAX_REQUESTS, retry: TOOL_RANK_MAX_REQUESTS + TOOL_RANK_RETRIES, }), rankOptions?.signal, ); const unscored = new Set(documents.map((document) => document.name)); const position = new Map(order.map((name, index) => [name, index])); const scored: ToolSearchMatch[] = []; items.forEach((item, index) => { const score = outcome.scores[index]; if (typeof score !== "number" || !Number.isFinite(score)) return; unscored.delete(item.key); if (score >= TOOL_RANK_KEEP) scored.push({ name: item.key, score }); }); if (unscored.size === documents.length) { telemetry("fallback", { confidence: confidenceBucket(undefined), reason: outcome.reasons[0] ?? "missing" }); return local.slice(0, limit); } scored.sort((a, b) => b.score - a.score || (position.get(a.name) ?? 0) - (position.get(b.name) ?? 0)); const extras = local.filter((match) => unscored.has(match.name)).map((match) => ({ name: match.name, score: 0 })); telemetry("jev", { judged: items.length, matched: scored.length, confidence: confidenceBucket(scored[0]?.score), ...(unscored.size > 0 ? { unscored: unscored.size } : {}), }); return [...scored, ...extras].slice(0, limit); } catch { // Discovery must work without Jev: an abort, a transport error, or a bad answer ranks locally. telemetry("fallback", { confidence: confidenceBucket(undefined), reason: "transport" }); return local.slice(0, limit); } }, }; } export default function jevToolRankerExtension(pi: ExtensionAPI): void { const dispose = registerToolRanker(createJevToolRanker({ events: pi.events })); pi.on("session_shutdown", () => { dispose(); }); }