s1 = soundgen(sylLen = 100, pitch = c(80, 180), formants = 'a')
s2 = soundgen(sylLen = 120, pitch = c(150, 350), formants = 'u')
compareSounds(s1, s2, samplingRate = 16000, method = c('cor', 'cosine', 'diff'))
# spectrogram(s1); playme(s1)
# spectrogram(s2); playme(s2)
if (FALSE) {
# NB: install the "dtw" library to run the examples
# compare all sounds in a folder, e.g.:
target = '~/Documents/Research/zz_test_audio/temp_long'
cf = compareFolder(target, logSpec = TRUE)
mds = as.data.frame(cmdscale(cf$cor))
plot(mds, type = 'n'); text(mds, labels = abbreviate(rownames(mds)))
# or use manually produced spectrograms
sp = spectrogram(target, windowLength = c(10, 40), overlap = 75,
yScale = 'ERB', output = 'processed', plot = FALSE, cores = 4)
image(sp[[1]])
cf1 = compareFolder(spectrograms = sp)
mds1 = as.data.frame(cmdscale(cf1$cor))
plot(mds1, type = 'n'); text(mds1, labels = abbreviate(rownames(mds1)))
# extract a spectrogram-like representation using a custom function
# (e.g., full-resolution analytic envelopes instead of downsampled RMS)
compareSounds(s1, s2, samplingRate = 16000,
specFun = function(x) matrix(hilbert_approx(x)$envelope, nrow = 1))
# some more examples
s1 = soundgen(formants = 'a', play = TRUE)
s2 = soundgen(formants = 'ae', play = TRUE)
s3 = soundgen(formants = 'eae', sylLen = 700, play = TRUE)
s4 = runif(8000, -1, 1) # white noise
compareSounds(s1, s2, samplingRate = 16000)
compareSounds(s1, s4, samplingRate = 16000)
# the central section of s3 is more similar to s1 than is the beg/end of s3
compareSounds(s1, s3, samplingRate = 16000, padDir = 'left')
compareSounds(s1, s3, samplingRate = 16000, padDir = 'central')
# padding with 0 penalizes differences in duration, whereas padding with NA
# is like saying we only care about the overlapping part
compareSounds(s1, s4, samplingRate = 16000, padWith = 0)
compareSounds(s1, s4, samplingRate = 16000, padWith = NA)
# different types of spectrograms produce quite different results
compareSounds(s1, s3, samplingRate = 16000, specFun = 'stft')
compareSounds(s1, s3, samplingRate = 16000, specFun = 'melspec')
compareSounds(s1, s3, samplingRate = 16000, specFun = 'mfcc')
compareSounds(s1, s3, samplingRate = 16000, specFun = 'audSpec')
# pass additional control parameters to specFun and DTW
compareSounds(s1, s3, samplingRate = 16000,
specFun = 'melspec',
specFun_pars = list(nbands = 128),
dtw_pars = list(dist.method = "Manhattan"))
# use feature matrices instead of spectrograms
# (time in columns, features in rows)
a1 = t(as.matrix(analyze(s1, samplingRate = 16000)$detailed))
a1 = a1[4:nrow(a1), ]; a1[is.na(a1)] = 0 # don't use dur and time stamps
a2 = t(as.matrix(analyze(s2, samplingRate = 16000)$detailed))
a2 = a2[4:nrow(a2), ]; a2[is.na(a2)] = 0
a4 = t(as.matrix(analyze(s4, samplingRate = 16000)$detailed))
a4 = a4[4:nrow(a4), ]; a4[is.na(a4)] = 0
compareSounds(a1, a2, method = c('cosine', 'dtw'))
compareSounds(a1, a4, method = c('cosine', 'dtw'))
}
Run the code above in your browser using DataLab