mirror of
https://github.com/signalwire/freeswitch.git
synced 2026-07-24 21:22:09 +00:00
i've tested, now you can too
This commit is contained in:
+163
-45
@@ -22,8 +22,7 @@
|
||||
License for more details.
|
||||
|
||||
You should have received a copy of the GNU Lesser General Public License
|
||||
along with this program; if not, write to the Free Software
|
||||
Foundation, Inc., 675 Mass Ave, Cambridge, MA 02139, USA.
|
||||
along with this program; if not, see <http://www.gnu.org/licenses/>.
|
||||
*/
|
||||
|
||||
/*---------------------------------------------------------------------------*\
|
||||
@@ -38,7 +37,9 @@
|
||||
|
||||
#include "defines.h"
|
||||
#include "sine.h"
|
||||
#include "four1.h"
|
||||
#include "kiss_fft.h"
|
||||
|
||||
#define HPF_BETA 0.125
|
||||
|
||||
/*---------------------------------------------------------------------------*\
|
||||
|
||||
@@ -65,9 +66,10 @@ void hs_pitch_refinement(MODEL *model, COMP Sw[], float pmin, float pmax,
|
||||
|
||||
\*---------------------------------------------------------------------------*/
|
||||
|
||||
void make_analysis_window(float w[],COMP W[])
|
||||
void make_analysis_window(kiss_fft_cfg fft_fwd_cfg, float w[], COMP W[])
|
||||
{
|
||||
float m;
|
||||
COMP wshift[FFT_ENC];
|
||||
COMP temp;
|
||||
int i,j;
|
||||
|
||||
@@ -122,15 +124,15 @@ void make_analysis_window(float w[],COMP W[])
|
||||
*/
|
||||
|
||||
for(i=0; i<FFT_ENC; i++) {
|
||||
W[i].real = 0.0;
|
||||
W[i].imag = 0.0;
|
||||
wshift[i].real = 0.0;
|
||||
wshift[i].imag = 0.0;
|
||||
}
|
||||
for(i=0; i<NW/2; i++)
|
||||
W[i].real = w[i+M/2];
|
||||
wshift[i].real = w[i+M/2];
|
||||
for(i=FFT_ENC-NW/2,j=M/2-NW/2; i<FFT_ENC; i++,j++)
|
||||
W[i].real = w[j];
|
||||
wshift[i].real = w[j];
|
||||
|
||||
four1(&W[-1].imag,FFT_ENC,-1); /* "Numerical Recipes in C" FFT */
|
||||
kiss_fft(fft_fwd_cfg, (kiss_fft_cpx *)wshift, (kiss_fft_cpx *)W);
|
||||
|
||||
/*
|
||||
Re-arrange W[] to be symmetrical about FFT_ENC/2. Makes later
|
||||
@@ -167,6 +169,26 @@ void make_analysis_window(float w[],COMP W[])
|
||||
|
||||
}
|
||||
|
||||
/*---------------------------------------------------------------------------*\
|
||||
|
||||
FUNCTION....: hpf
|
||||
AUTHOR......: David Rowe
|
||||
DATE CREATED: 16 Nov 2010
|
||||
|
||||
High pass filter with a -3dB point of about 160Hz.
|
||||
|
||||
y(n) = -HPF_BETA*y(n-1) + x(n) - x(n-1)
|
||||
|
||||
\*---------------------------------------------------------------------------*/
|
||||
|
||||
float hpf(float x, float states[])
|
||||
{
|
||||
states[0] += -HPF_BETA*states[0] + x - states[1];
|
||||
states[1] = x;
|
||||
|
||||
return states[0];
|
||||
}
|
||||
|
||||
/*---------------------------------------------------------------------------*\
|
||||
|
||||
FUNCTION....: dft_speech
|
||||
@@ -177,13 +199,14 @@ void make_analysis_window(float w[],COMP W[])
|
||||
|
||||
\*---------------------------------------------------------------------------*/
|
||||
|
||||
void dft_speech(COMP Sw[], float Sn[], float w[])
|
||||
void dft_speech(kiss_fft_cfg fft_fwd_cfg, COMP Sw[], float Sn[], float w[])
|
||||
{
|
||||
int i;
|
||||
|
||||
int i;
|
||||
COMP sw[FFT_ENC];
|
||||
|
||||
for(i=0; i<FFT_ENC; i++) {
|
||||
Sw[i].real = 0.0;
|
||||
Sw[i].imag = 0.0;
|
||||
sw[i].real = 0.0;
|
||||
sw[i].imag = 0.0;
|
||||
}
|
||||
|
||||
/* Centre analysis window on time axis, we need to arrange input
|
||||
@@ -192,14 +215,14 @@ void dft_speech(COMP Sw[], float Sn[], float w[])
|
||||
/* move 2nd half to start of FFT input vector */
|
||||
|
||||
for(i=0; i<NW/2; i++)
|
||||
Sw[i].real = Sn[i+M/2]*w[i+M/2];
|
||||
sw[i].real = Sn[i+M/2]*w[i+M/2];
|
||||
|
||||
/* move 1st half to end of FFT input vector */
|
||||
|
||||
for(i=0; i<NW/2; i++)
|
||||
Sw[FFT_ENC-NW/2+i].real = Sn[i+M/2-NW/2]*w[i+M/2-NW/2];
|
||||
sw[FFT_ENC-NW/2+i].real = Sn[i+M/2-NW/2]*w[i+M/2-NW/2];
|
||||
|
||||
four1(&Sw[-1].imag,FFT_ENC,-1);
|
||||
kiss_fft(fft_fwd_cfg, (kiss_fft_cpx *)sw, (kiss_fft_cpx *)Sw);
|
||||
}
|
||||
|
||||
/*---------------------------------------------------------------------------*\
|
||||
@@ -362,38 +385,38 @@ float est_voicing_mbe(
|
||||
MODEL *model,
|
||||
COMP Sw[],
|
||||
COMP W[],
|
||||
float f0,
|
||||
COMP Sw_[] /* DFT of all voiced synthesised signal for f0 */
|
||||
/* useful for debugging/dump file */
|
||||
)
|
||||
COMP Sw_[], /* DFT of all voiced synthesised signal */
|
||||
/* useful for debugging/dump file */
|
||||
COMP Ew[], /* DFT of error */
|
||||
float prev_Wo)
|
||||
{
|
||||
int i,l,al,bl,m; /* loop variables */
|
||||
COMP Am; /* amplitude sample for this band */
|
||||
int offset; /* centers Hw[] about current harmonic */
|
||||
float den; /* denominator of Am expression */
|
||||
float error; /* accumulated error between originl and synthesised */
|
||||
float Wo; /* current "test" fundamental freq. */
|
||||
int L;
|
||||
float error; /* accumulated error between original and synthesised */
|
||||
float Wo;
|
||||
float sig, snr;
|
||||
float elow, ehigh, eratio;
|
||||
float dF0, sixty;
|
||||
|
||||
sig = 0.0;
|
||||
sig = 1E-4;
|
||||
for(l=1; l<=model->L/4; l++) {
|
||||
sig += model->A[l]*model->A[l];
|
||||
}
|
||||
|
||||
for(i=0; i<FFT_ENC; i++) {
|
||||
Sw_[i].real = 0.0;
|
||||
Sw_[i].imag = 0.0;
|
||||
Ew[i].real = 0.0;
|
||||
Ew[i].imag = 0.0;
|
||||
}
|
||||
|
||||
L = floor((FS/2.0)/f0);
|
||||
Wo = f0*(TWO_PI/FS);
|
||||
|
||||
error = 0.0;
|
||||
Wo = model->Wo;
|
||||
error = 1E-4;
|
||||
|
||||
/* Just test across the harmonics in the first 1000 Hz (L/4) */
|
||||
|
||||
for(l=1; l<=L/4; l++) {
|
||||
for(l=1; l<=model->L/4; l++) {
|
||||
Am.real = 0.0;
|
||||
Am.imag = 0.0;
|
||||
den = 0.0;
|
||||
@@ -418,16 +441,73 @@ float est_voicing_mbe(
|
||||
offset = FFT_ENC/2 + m - l*Wo*FFT_ENC/TWO_PI + 0.5;
|
||||
Sw_[m].real = Am.real*W[offset].real - Am.imag*W[offset].imag;
|
||||
Sw_[m].imag = Am.real*W[offset].imag + Am.imag*W[offset].real;
|
||||
error += (Sw[m].real - Sw_[m].real)*(Sw[m].real - Sw_[m].real);
|
||||
error += (Sw[m].imag - Sw_[m].imag)*(Sw[m].imag - Sw_[m].imag);
|
||||
Ew[m].real = Sw[m].real - Sw_[m].real;
|
||||
Ew[m].imag = Sw[m].imag - Sw_[m].imag;
|
||||
error += Ew[m].real*Ew[m].real;
|
||||
error += Ew[m].imag*Ew[m].imag;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
snr = 10.0*log10(sig/error);
|
||||
if (snr > V_THRESH)
|
||||
model->voiced = 1;
|
||||
else
|
||||
model->voiced = 0;
|
||||
|
||||
/* post processing, helps clean up some voicing errors ------------------*/
|
||||
|
||||
/*
|
||||
Determine the ratio of low freqency to high frequency energy,
|
||||
voiced speech tends to be dominated by low frequency energy,
|
||||
unvoiced by high frequency. This measure can be used to
|
||||
determine if we have made any gross errors.
|
||||
*/
|
||||
|
||||
elow = ehigh = 1E-4;
|
||||
for(l=1; l<=model->L/2; l++) {
|
||||
elow += model->A[l]*model->A[l];
|
||||
}
|
||||
for(l=model->L/2; l<=model->L; l++) {
|
||||
ehigh += model->A[l]*model->A[l];
|
||||
}
|
||||
eratio = 10.0*log10(elow/ehigh);
|
||||
dF0 = 0.0;
|
||||
|
||||
/* Look for Type 1 errors, strongly V speech that has been
|
||||
accidentally declared UV */
|
||||
|
||||
if (model->voiced == 0)
|
||||
if (eratio > 10.0)
|
||||
model->voiced = 1;
|
||||
|
||||
/* Look for Type 2 errors, strongly UV speech that has been
|
||||
accidentally declared V */
|
||||
|
||||
if (model->voiced == 1) {
|
||||
if (eratio < -10.0)
|
||||
model->voiced = 0;
|
||||
|
||||
/* If pitch is jumping about it's likely this is UV */
|
||||
|
||||
/* 13 Feb 2012 - this seems to add some V errors so comment out for now. Maybe
|
||||
double check on bg noise files
|
||||
|
||||
dF0 = (model->Wo - prev_Wo)*FS/TWO_PI;
|
||||
if (fabs(dF0) > 15.0)
|
||||
model->voiced = 0;
|
||||
*/
|
||||
|
||||
/* A common source of Type 2 errors is the pitch estimator
|
||||
gives a low (50Hz) estimate for UV speech, which gives a
|
||||
good match with noise due to the close harmoonic spacing.
|
||||
These errors are much more common than people with 50Hz3
|
||||
pitch, so we have just a small eratio threshold. */
|
||||
|
||||
sixty = 60.0*TWO_PI/FS;
|
||||
if ((eratio < -4.0) && (model->Wo <= sixty))
|
||||
model->voiced = 0;
|
||||
}
|
||||
//printf(" v: %d snr: %f eratio: %3.2f %f\n",model->voiced,snr,eratio,dF0);
|
||||
|
||||
return snr;
|
||||
}
|
||||
@@ -471,20 +551,22 @@ void make_synthesis_window(float Pn[])
|
||||
DATE CREATED: 20/2/95
|
||||
|
||||
Synthesise a speech signal in the frequency domain from the
|
||||
sinusodal model parameters. Uses overlap-add a triangular window to
|
||||
smoothly interpolate betwen frames.
|
||||
sinusodal model parameters. Uses overlap-add with a trapezoidal
|
||||
window to smoothly interpolate betwen frames.
|
||||
|
||||
\*---------------------------------------------------------------------------*/
|
||||
|
||||
void synthesise(
|
||||
float Sn_[], /* time domain synthesised signal */
|
||||
MODEL *model, /* ptr to model parameters for this frame */
|
||||
float Pn[], /* time domain Parzen window */
|
||||
int shift /* used to handle transition frames */
|
||||
kiss_fft_cfg fft_inv_cfg,
|
||||
float Sn_[], /* time domain synthesised signal */
|
||||
MODEL *model, /* ptr to model parameters for this frame */
|
||||
float Pn[], /* time domain Parzen window */
|
||||
int shift /* flag used to handle transition frames */
|
||||
)
|
||||
{
|
||||
int i,l,j,b; /* loop variables */
|
||||
COMP Sw_[FFT_DEC]; /* DFT of synthesised signal */
|
||||
COMP sw_[FFT_DEC]; /* synthesised signal */
|
||||
|
||||
if (shift) {
|
||||
/* Update memories */
|
||||
@@ -500,10 +582,30 @@ void synthesise(
|
||||
Sw_[i].imag = 0.0;
|
||||
}
|
||||
|
||||
/* Now set up frequency domain synthesised speech */
|
||||
/*
|
||||
Nov 2010 - found that synthesis using time domain cos() functions
|
||||
gives better results for synthesis frames greater than 10ms. Inverse
|
||||
FFT synthesis using a 512 pt FFT works well for 10ms window. I think
|
||||
(but am not sure) that the problem is related to the quantisation of
|
||||
the harmonic frequencies to the FFT bin size, e.g. there is a
|
||||
8000/512 Hz step between FFT bins. For some reason this makes
|
||||
the speech from longer frame > 10ms sound poor. The effect can also
|
||||
be seen when synthesising test signals like single sine waves, some
|
||||
sort of amplitude modulation at the frame rate.
|
||||
|
||||
Another possibility is using a larger FFT size (1024 or 2048).
|
||||
*/
|
||||
|
||||
#define FFT_SYNTHESIS
|
||||
#ifdef FFT_SYNTHESIS
|
||||
/* Now set up frequency domain synthesised speech */
|
||||
for(l=1; l<=model->L; l++) {
|
||||
//for(l=model->L/2; l<=model->L; l++) {
|
||||
//for(l=1; l<=model->L/4; l++) {
|
||||
b = floor(l*model->Wo*FFT_DEC/TWO_PI + 0.5);
|
||||
if (b > ((FFT_DEC/2)-1)) {
|
||||
b = (FFT_DEC/2)-1;
|
||||
}
|
||||
Sw_[b].real = model->A[l]*cos(model->phi[l]);
|
||||
Sw_[b].imag = model->A[l]*sin(model->phi[l]);
|
||||
Sw_[FFT_DEC-b].real = Sw_[b].real;
|
||||
@@ -512,19 +614,35 @@ void synthesise(
|
||||
|
||||
/* Perform inverse DFT */
|
||||
|
||||
four1(&Sw_[-1].imag,FFT_DEC,1);
|
||||
kiss_fft(fft_inv_cfg, (kiss_fft_cpx *)Sw_, (kiss_fft_cpx *)sw_);
|
||||
#else
|
||||
/*
|
||||
Direct time domain synthesis using the cos() function. Works
|
||||
well at 10ms and 20ms frames rates. Note synthesis window is
|
||||
still used to handle overlap-add between adjacent frames. This
|
||||
could be simplified as we don't need to synthesise where Pn[]
|
||||
is zero.
|
||||
*/
|
||||
for(l=1; l<=model->L; l++) {
|
||||
for(i=0,j=-N+1; i<N-1; i++,j++) {
|
||||
Sw_[FFT_DEC-N+1+i].real += 2.0*model->A[l]*cos(j*model->Wo*l + model->phi[l]);
|
||||
}
|
||||
for(i=N-1,j=0; i<2*N; i++,j++)
|
||||
Sw_[j].real += 2.0*model->A[l]*cos(j*model->Wo*l + model->phi[l]);
|
||||
}
|
||||
#endif
|
||||
|
||||
/* Overlap add to previous samples */
|
||||
|
||||
for(i=0; i<N-1; i++) {
|
||||
Sn_[i] += Sw_[FFT_DEC-N+1+i].real*Pn[i];
|
||||
Sn_[i] += sw_[FFT_DEC-N+1+i].real*Pn[i];
|
||||
}
|
||||
|
||||
if (shift)
|
||||
for(i=N-1,j=0; i<2*N; i++,j++)
|
||||
Sn_[i] = Sw_[j].real*Pn[i];
|
||||
Sn_[i] = sw_[j].real*Pn[i];
|
||||
else
|
||||
for(i=N-1,j=0; i<2*N; i++,j++)
|
||||
Sn_[i] += Sw_[j].real*Pn[i];
|
||||
Sn_[i] += sw_[j].real*Pn[i];
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user