/** * VBx clustering. Direct port of pyannote.audio/utils/vbx.py (4.0.4), * which itself is derived from BUT VBx (Landini et al., 2022) without HMM. * * Notation matches the original Python; comments reference equation numbers * from the paper "Bayesian HMM clustering of x-vector sequences (VBx)". */ import type { Mat } from "../math/linalg.js"; export interface VBxResult { /** (T, S) responsibilities (marginal posteriors per frame and speaker). */ gamma: Mat; /** (S,) ML-learned speaker priors. Speakers with pi ~ 0 were pruned by the model. */ pi: Float64Array; /** ELBO trajectory (one entry per VB iteration). */ elbo: number[]; } export interface VBxOptions { Fa?: number; Fb?: number; maxIters?: number; epsilon?: number; /** AHC initialization labels, length T (values in 0..K-1). */ ahcInit: Int32Array; /** Smoothing applied to one-hot AHC init via softmax(qinit * smoothing). Use < 0 to disable. */ initSmoothing?: number; } /** * VBx core algorithm. PLDA assumes zero mean, diagonal across-class covariance Phi, * identity within-class covariance. * * X - (T, D) PLDA-transformed feature vectors (one per chunk-speaker) * Phi - (D,) between-class diagonal covariance (= plda.psi[:lda_dim]) */ export declare function vbx(X: Mat, Phi: Float64Array, opts: VBxOptions): VBxResult; /** * pyannote-style cluster_vbx wrapper. * Returns gamma (T, S) and pi (S,). Speakers with pi <= 1e-7 should be discarded. */ export declare function clusterVbx(fea: Mat, phi: Float64Array, ahcInit: Int32Array, options: { Fa: number; Fb: number; maxIters?: number; initSmoothing?: number; }): VBxResult; //# sourceMappingURL=vbx.d.ts.map