export interface FbankConfig { numMelBins: number; frameLengthMs: number; frameShiftMs: number; sampleFrequency: number; /** "hamming" only for now (matches WeSpeaker / pyannote). */ windowType: "hamming"; preemphasisCoefficient: number; removeDcOffset: boolean; roundToPowerOfTwo: boolean; lowFreq: number; /** 0 means Nyquist (sampleFrequency / 2). */ highFreq: number; usePower: boolean; useLogFbank: boolean; /** Float32 epsilon, used by torchaudio for the log floor. */ epsilon: number; } export declare const PYANNOTE_FBANK_CONFIG: FbankConfig; /** * Kaldi-compatible fbank features. Mirrors torchaudio.compliance.kaldi.fbank * with the parameter set used by the WeSpeaker / pyannote pipeline. * * Returns row-major (numFrames, numMelBins) as a single Float64Array, plus shape. */ export declare function kaldiFbank(waveform: Float32Array | Float64Array, config?: FbankConfig): { data: Float64Array; numFrames: number; numBins: number; }; /** * Pyannote / WeSpeaker fbank: Kaldi fbank + per-bin mean normalization across time. * This is what the embedding ResNet34 expects as input. */ export declare function pyannoteFbank(waveform: Float32Array | Float64Array, config?: FbankConfig): { data: Float64Array; numFrames: number; numBins: number; }; //# sourceMappingURL=fbank.d.ts.map