/** * Pitch tracking using parabolic interpolation of peak locations in a spectrogram * @param {Float32Array} y - Audio time series (optional if S provided) * @param {number} sr - Sample rate * @param {Array} S - Pre-computed magnitude/power spectrogram [freq][time] * @param {number} n_fft - FFT window size * @param {number} hop_length - Hop length * @param {number} fmin - Minimum frequency * @param {number} fmax - Maximum frequency * @param {number} threshold - Threshold for peak detection * @returns {Object} {pitches: Array, magnitudes: Array} - pitch and magnitude per frame */ export function piptrack(y?: Float32Array, sr?: number, S?: any[], n_fft?: number, hop_length?: number, fmin?: number, fmax?: number, threshold?: number): any; /** * Fundamental frequency (F0) estimation using the YIN algorithm * @param {Float32Array} y - Audio time series * @param {number} fmin - Minimum frequency to search * @param {number} fmax - Maximum frequency to search * @param {number} sr - Sample rate * @param {number} frame_length - Length of analysis frame * @param {number} win_length - Window length (default: frame_length / 2) * @param {number} hop_length - Hop length * @param {number} trough_threshold - Threshold for peak picking * @returns {Float32Array} F0 estimates per frame (0 = unvoiced) */ export function yin(y: Float32Array, fmin?: number, fmax?: number, sr?: number, frame_length?: number, win_length?: number, hop_length?: number, trough_threshold?: number): Float32Array; /** * Probabilistic YIN (pYIN). * * Two-stage pYIN pipeline: * 1. YIN cumulative-mean-normalized-difference per frame → local minima * (troughs) below a beta-distributed threshold ensemble, each weighted by * a Boltzmann prior over trough rank and the beta pmf over thresholds → * an observation matrix over a log-spaced f0 grid (n_bins_per_semitone * bins/semitone) stacked with an unvoiced state block. * 2. Transition matrix = transition_local band over the pitch grid ⊗ * voiced/unvoiced switching (transition_loop(2, 1 - switch_prob)) via a * Kronecker product np.kron(t_switch, transition). * 3. sequence.viterbi decode → per-frame pitch bin → f0 (fill_na when * unvoiced), voiced_flag, voiced_prob. * * Validated against committed reference fixtures (220→330 Hz step + silent * tail; voiced f0 within ~1 semitone, voicing exact on the clearly voiced/ * silent regions). This is the real pYIN — NOT the former median-over- * threshold-ensemble stub (no transition matrix, no Viterbi) that was honestly * left unexported. * * @param {Float32Array|number[]} y - Audio time series. * @param {number} fmin - Minimum frequency (Hz), > 0. * @param {number} fmax - Maximum frequency (Hz), fmin < fmax <= sr/2. * @param {number} [sr=22050] - Sample rate (Hz). * @param {object} [opts] * @param {number} [opts.frame_length=2048] * @param {number|null} [opts.hop_length=null] - Defaults to frame_length/4. * @param {number} [opts.n_thresholds=100] - Threshold-ensemble size. * @param {[number,number]} [opts.beta_parameters=[2,18]] - Beta prior (a, b). * @param {number} [opts.boltzmann_parameter=2] - Boltzmann prior over troughs. * @param {number} [opts.resolution=0.1] - Pitch-bin resolution in semitones. * @param {number} [opts.max_transition_rate=35.92] - Max transition (oct/sec). * @param {number} [opts.switch_prob=0.01] - Voiced↔unvoiced switch prob. * @param {number} [opts.no_trough_prob=0.01] - Best-guess mass when no trough. * @param {number} [opts.fill_na=NaN] - Value written to unvoiced f0 frames. * @param {boolean} [opts.center=true] - Center-pad frames (default). * @returns {{ f0: Float64Array, voiced_flag: boolean[], voiced_prob: Float64Array }} */ export function pyin(y: Float32Array | number[], fmin: number, fmax: number, sr?: number, { frame_length, hop_length, n_thresholds, beta_parameters, boltzmann_parameter, resolution, max_transition_rate, switch_prob, no_trough_prob, fill_na, center, }?: { frame_length?: number; hop_length?: number | null; n_thresholds?: number; beta_parameters?: [number, number]; boltzmann_parameter?: number; resolution?: number; max_transition_rate?: number; switch_prob?: number; no_trough_prob?: number; fill_na?: number; center?: boolean; }): { f0: Float64Array; voiced_flag: boolean[]; voiced_prob: Float64Array; }; /** * Estimate pitch using autocorrelation method * @param {Float32Array} y - Audio time series * @param {number} sr - Sample rate * @param {number} fmin - Minimum frequency * @param {number} fmax - Maximum frequency * @param {number} frame_length - Frame length * @param {number} hop_length - Hop length * @returns {Float32Array} F0 estimates per frame */ export function autocorrelation_pitch(y: Float32Array, sr?: number, fmin?: number, fmax?: number, frame_length?: number, hop_length?: number): Float32Array; /** * Convert pitch (Hz) to MIDI note number * @param {Float32Array|Array} pitches - Pitches in Hz * @returns {Float32Array} MIDI note numbers */ export function hz_to_midi_pitch(pitches: Float32Array | any[]): Float32Array; /** * Estimate pitch salience (confidence) * @param {Float32Array} y - Audio time series * @param {Float32Array} f0 - F0 estimates * @param {number} sr - Sample rate * @param {number} hop_length - Hop length * @returns {Float32Array} Salience values [0, 1] */ export function pitch_salience(y: Float32Array, f0: Float32Array, sr?: number, hop_length?: number): Float32Array; /** * Smooth pitch contour using median filtering * @param {Float32Array} f0 - F0 estimates * @param {number} window_size - Median filter window size (odd number) * @returns {Float32Array} Smoothed F0 */ export function smooth_pitch(f0: Float32Array, window_size?: number): Float32Array; /** * Given a collection of pitches, estimate its tuning offset (in fractions of a bin) * * This function estimates the deviation from 12-tone equal temperament (12-TET) * by analyzing the distribution of pitch deviations from semitone centers. * * @param {Array|Float32Array} frequencies - Collection of frequencies in Hz * @param {number} resolution - Resolution of tuning offset (default: 0.01 semitones) * @param {number} bins_per_octave - Number of bins per octave (default: 12 for semitones) * @returns {number} Tuning offset in fractions of bins_per_octave * * @example * // If frequencies are tuned 0.2 semitones sharp * pitch_tuning([442, 496, 590]) // ~0.2 */ export function pitch_tuning(frequencies: any[] | Float32Array, resolution?: number, bins_per_octave?: number): number; /** * Estimate the tuning of an audio time series or spectrogram input * * @param {Float32Array} y - Audio time series (optional if S provided) * @param {number} sr - Sample rate (default: 22050) * @param {Array} S - Spectrogram (optional if y provided) * @param {number} n_fft - FFT window size (default: 2048) * @param {number} resolution - Resolution of tuning offset (default: 0.01) * @param {number} bins_per_octave - Number of bins per octave (default: 12) * @param {Object} kwargs - Additional arguments passed to piptrack * @returns {number} Tuning deviation from A440 in fractions of bins_per_octave * * @example * const tuning = estimate_tuning(audioData, 22050) * console.log(`Audio is ${tuning * 100} cents sharp`) */ export function estimate_tuning(y?: Float32Array, sr?: number, S?: any[], n_fft?: number, resolution?: number, bins_per_octave?: number, kwargs?: any): number;