/* Copyright (c) 2011 Xiph.Org Foundation
Written by Jean-Marc Valin */ /* Redistributionanduseinsourceandbinaryforms,withorwithout modification,arepermittedprovidedthatthefollowingconditions aremet:
static opus_val32 silk_resampler_down2_hp(
opus_val32 *S, /* I/O State vector [ 2 ] */
opus_val32 *out, /* O Output signal [ floor(len/2) ] */ const opus_val32 *in, /* I Input signal [ len ] */ int inLen /* I Number of input samples */
)
{ int k, len2 = inLen/2;
opus_val32 in32, out32, out32_hp, Y, X;
opus_val64 hp_ener = 0; /* Internal variables and state are in Q10 format */ for( k = 0; k < len2; k++ ) { /* Convert to Q10 */
in32 = in[ 2 * k ];
/* All-pass section for even input sample */
Y = SUB32( in32, S[ 0 ] );
X = MULT16_32_Q15(QCONST16(0.6074371f, 15), Y);
out32 = ADD32( S[ 0 ], X );
S[ 0 ] = ADD32( in32, X );
out32_hp = out32; /* Convert to Q10 */
in32 = in[ 2 * k + 1 ];
/* All-pass section for odd input sample, and add to output of previous section */
Y = SUB32( in32, S[ 1 ] );
X = MULT16_32_Q15(QCONST16(0.15063f, 15), Y);
out32 = ADD32( out32, S[ 1 ] );
out32 = ADD32( out32, X );
S[ 1 ] = ADD32( in32, X );
Y = SUB32( -in32, S[ 2 ] );
X = MULT16_32_Q15(QCONST16(0.15063f, 15), Y);
out32_hp = ADD32( out32_hp, S[ 2 ] );
out32_hp = ADD32( out32_hp, X );
S[ 2 ] = ADD32( -in32, X );
/* len2 can be up to 480, so we shift by 8 to make it fit. */
hp_ener += SHR64(out32_hp*(opus_val64)out32_hp, 8); /* Add, convert back to int16 and store to output */
out[ k ] = HALF32(out32);
} #ifdef FIXED_POINT /* Fitting in 32 bits. */
hp_ener = hp_ener >> (2*SIG_SHIFT); if (hp_ener > 2147483647) hp_ener = 2147483647; #endif return (opus_val32)hp_ener;
}
static opus_val32 downmix_and_resample(downmix_func downmix, constvoid *_x, opus_val32 *y, opus_val32 S[3], int subframe, int offset, int c1, int c2, int C, int Fs)
{
VARDECL(opus_val32, tmp); int j;
opus_val32 ret = 0;
SAVE_STACK;
void tonality_get_info(TonalityAnalysisState *tonal, AnalysisInfo *info_out, int len)
{ int pos; int curr_lookahead; float tonality_max; float tonality_avg; int tonality_count; int i; int pos0; float prob_avg; float prob_count; float prob_min, prob_max; float vad_prob; int mpos, vpos; int bandwidth_span;
tonal->read_subframe += len/(tonal->Fs/400); while (tonal->read_subframe>=8)
{
tonal->read_subframe -= 8;
tonal->read_pos++;
} if (tonal->read_pos>=DETECT_SIZE)
tonal->read_pos-=DETECT_SIZE;
/* On long frames, look at the second analysis window rather than the first. */ if (len > tonal->Fs/50 && pos != tonal->write_pos)
{
pos++; if (pos==DETECT_SIZE)
pos=0;
} if (pos == tonal->write_pos)
pos--; if (pos<0)
pos = DETECT_SIZE-1;
pos0 = pos;
OPUS_COPY(info_out, &tonal->info[pos], 1); if (!info_out->valid) return;
tonality_max = tonality_avg = info_out->tonality;
tonality_count = 1; /* Look at the neighbouring frames and pick largest bandwidth found (to be safe). */
bandwidth_span = 6; /* If possible, look ahead for a tone to compensate for the delay in the tone detector. */ for (i=0;i<3;i++)
{
pos++; if (pos==DETECT_SIZE)
pos = 0; if (pos == tonal->write_pos) break;
tonality_max = MAX32(tonality_max, tonal->info[pos].tonality);
tonality_avg += tonal->info[pos].tonality;
tonality_count++;
info_out->bandwidth = IMAX(info_out->bandwidth, tonal->info[pos].bandwidth);
bandwidth_span--;
}
pos = pos0; /* Look back in time to see if any has a wider bandwidth than the current frame. */ for (i=0;i<bandwidth_span;i++)
{
pos--; if (pos < 0)
pos = DETECT_SIZE-1; if (pos == tonal->write_pos) break;
info_out->bandwidth = IMAX(info_out->bandwidth, tonal->info[pos].bandwidth);
}
info_out->tonality = MAX32(tonality_avg/tonality_count, tonality_max-.2f);
mpos = vpos = pos0; /* If we have enough look-ahead, compensate for the ~5-frame delay in the music prob and
~1 frame delay in the VAD prob. */ if (curr_lookahead > 15)
{
mpos += 5; if (mpos>=DETECT_SIZE)
mpos -= DETECT_SIZE;
vpos += 1; if (vpos>=DETECT_SIZE)
vpos -= DETECT_SIZE;
}
/* The following calculations attempt to minimize a "badness function" forthetransition.Whenswitchingfromspeechtomusic,thebadness ofswitchingatframekis b_k=S*v_k+\sum_{i=0}^{k-1}v_i*(p_i-T) where v_iistheactivityprobability(VAD)atframei, p_iisthemusicprobabilityatframei Tistheprobabilitythresholdforswitching Sisthepenaltyforswitchingduringactiveaudioratherthansilence thecurrentframehasindexi=0
#ifdef FIXED_POINT /* For fixed-point, the input is +/-2^15 shifted up by SIG_SHIFT, so we need to
compensate for that in the energy. */ #define SCALE_COMPENS (1.f/((opus_int32)1<<(15+SIG_SHIFT))) #define SCALE_ENER(e) ((SCALE_COMPENS*SCALE_COMPENS)*(e)) #else #define SCALE_ENER(e) ((1.f/32768/32768)*e) #endif
#ifdef FIXED_POINT staticint is_digital_silence32(const opus_val32* pcm, int frame_size, int channels, int lsb_depth)
{ int silence = 0;
opus_val32 sample_max = 0; #ifdef MLP_TRAINING return0; #endif
sample_max = celt_maxabs32(pcm, frame_size*channels);
staticvoid tonality_analysis(TonalityAnalysisState *tonal, const CELTMode *celt_mode, constvoid *x, int len, int offset, int c1, int c2, int C, int lsb_depth, downmix_func downmix)
{ int i, b; const kiss_fft_state *kfft;
VARDECL(kiss_fft_cpx, in);
VARDECL(kiss_fft_cpx, out); int N = 480, N2=240; float * OPUS_RESTRICT A = tonal->angle; float * OPUS_RESTRICT dA = tonal->d_angle; float * OPUS_RESTRICT d2A = tonal->d2_angle;
VARDECL(float, tonality);
VARDECL(float, noisiness); float band_tonality[NB_TBANDS]; float logE[NB_TBANDS]; float BFCC[8]; float features[25]; float frame_tonality; float max_frame_tonality; /*float tw_sum=0;*/ float frame_noisiness; constfloat pi4 = (float)(M_PI*M_PI*M_PI*M_PI); float slope=0; float frame_stationarity; float relativeE; float frame_probs[2]; float alpha, alphaE, alphaE2; float frame_loudness; float bandwidth_mask; int is_masked[NB_TBANDS+1]; int bandwidth=0; float maxE = 0; float noise_floor; int remaining;
AnalysisInfo *info; float hp_ener; float tonality2[240]; float midE[8]; float spec_variability=0; float band_log2[NB_TBANDS+1]; float leakage_from[NB_TBANDS+1]; float leakage_to[NB_TBANDS+1]; float layer_out[MAX_NEURONS]; float below_max_pitch; float above_max_pitch; int is_silence;
SAVE_STACK;
if (!tonal->initialized)
{
tonal->mem_fill = 240;
tonal->initialized = 1;
}
alpha = 1.f/IMIN(10, 1+tonal->count);
alphaE = 1.f/IMIN(25, 1+tonal->count); /* Noise floor related decay for bandwidth detection: -2.2 dB/second */
alphaE2 = 1.f/IMIN(100, 1+tonal->count); if (tonal->count <= 1) alphaE2 = 1;
if (tonal->Fs == 48000)
{ /* len and offset are now at 24 kHz. */
len/= 2;
offset /= 2;
} elseif (tonal->Fs == 16000) {
len = 3*len/2;
offset = 3*offset/2;
}
kfft = celt_mode->mdct.kfft[0];
tonal->hp_ener_accum += (float)downmix_and_resample(downmix, x,
&tonal->inmem[tonal->mem_fill], tonal->downmix_state,
IMIN(len, ANALYSIS_BUF_SIZE-tonal->mem_fill), offset, c1, c2, C, tonal->Fs); if (tonal->mem_fill+len < ANALYSIS_BUF_SIZE)
{
tonal->mem_fill += len; /* Don't have enough to update the analysis */
RESTORE_STACK; return;
}
hp_ener = tonal->hp_ener_accum;
info = &tonal->info[tonal->write_pos++]; if (tonal->write_pos>=DETECT_SIZE)
tonal->write_pos-=DETECT_SIZE;
avg_mod = .25f*(d2A[i]+mod1+2*mod2); /* This introduces an extra delay of 2 frames in the detection. */
tonality[i] = 1.f/(1.f+40.f*16.f*pi4*avg_mod)-.015f; /* No delay on this detection, but it's less reliable. */
tonality2[i] = 1.f/(1.f+40.f*16.f*pi4*mod2)-.015f;
A[i] = angle2;
dA[i] = d_angle2;
d2A[i] = mod2;
} for (i=2;i<N2-1;i++)
{ float tt = MIN32(tonality2[i], MAX32(tonality2[i-1], tonality2[i+1]));
tonality[i] = .9f*MAX32(tonality[i], tt-.1f);
}
frame_tonality = 0;
max_frame_tonality = 0; /*tw_sum = 0;*/
info->activity = 0;
frame_noisiness = 0;
frame_stationarity = 0; if (!tonal->count)
{ for (b=0;b<NB_TBANDS;b++)
{
tonal->lowE[b] = 1e10;
tonal->highE[b] = -1e10;
}
}
relativeE = 0;
frame_loudness = 0; /* The energy of the very first band is special because of DC. */
{ float E = 0; float X1r, X2r;
X1r = 2*(float)out[0].r;
X2r = 2*(float)out[0].i;
E = X1r*X1r + X2r*X2r; for (i=1;i<4;i++)
{ float binE = out[i].r*(float)out[i].r + out[N-i].r*(float)out[N-i].r
+ out[i].i*(float)out[i].i + out[N-i].i*(float)out[N-i].i;
E += binE;
}
E = SCALE_ENER(E);
band_log2[0] = .5f*1.442695f*(float)log(E+1e-10f);
} for (b=0;b<NB_TBANDS;b++)
{ float E=0, tE=0, nE=0; float L1, L2; float stationarity; for (i=tbands[b];i<tbands[b+1];i++)
{ float binE = out[i].r*(float)out[i].r + out[N-i].r*(float)out[N-i].r
+ out[i].i*(float)out[i].i + out[N-i].i*(float)out[N-i].i;
binE = SCALE_ENER(binE);
E += binE;
tE += binE*MAX32(0, tonality[i]);
nE += binE*2.f*(.5f-noisiness[i]);
} #ifndef FIXED_POINT /* Check for extreme band energies that could cause NaNs later. */ if (!(E<1e9f) || celt_isnan(E))
{
info->valid = 0;
RESTORE_STACK; return;
} #endif
leakage_from[0] = band_log2[0];
leakage_to[0] = band_log2[0] - LEAKAGE_OFFSET; for (b=1;b<NB_TBANDS+1;b++)
{ float leak_slope = LEAKAGE_SLOPE*(tbands[b]-tbands[b-1])/4;
leakage_from[b] = MIN16(leakage_from[b-1]+leak_slope, band_log2[b]);
leakage_to[b] = MAX16(leakage_to[b-1]-leak_slope, band_log2[b]-LEAKAGE_OFFSET);
} for (b=NB_TBANDS-2;b>=0;b--)
{ float leak_slope = LEAKAGE_SLOPE*(tbands[b+1]-tbands[b])/4;
leakage_from[b] = MIN16(leakage_from[b+1]+leak_slope, leakage_from[b]);
leakage_to[b] = MAX16(leakage_to[b+1]-leak_slope, leakage_to[b]);
}
celt_assert(NB_TBANDS+1 <= LEAK_BANDS); for (b=0;b<NB_TBANDS+1;b++)
{ /* leak_boost[] is made up of two terms. The first, based on leakage_to[], representstheboostneededtoovercometheamountofanalysisleakage causeinaweakerbandbbylouderneighbouringbands. Thesecond,basedonleakage_from[],appliestoaloudbandbfor whichthequantizationnoisecausessynthesisleakagetotheweaker
neighbouring bands. */ float boost = MAX16(0, leakage_to[b] - band_log2[b]) +
MAX16(0, band_log2[b] - (leakage_from[b]+LEAKAGE_OFFSET));
info->leak_boost[b] = IMIN(255, (int)floor(.5 + 64.f*boost));
} for (;b<LEAK_BANDS;b++) info->leak_boost[b] = 0;
for (i=0;i<NB_FRAMES;i++)
{ int j; float mindist = 1e15f; for (j=0;j<NB_FRAMES;j++)
{ int k; float dist=0; for (k=0;k<NB_TBANDS;k++)
{ float tmp;
tmp = tonal->logE[i][k] - tonal->logE[j][k];
dist += tmp*tmp;
} if (j!=i)
mindist = MIN32(mindist, dist);
}
spec_variability += mindist;
}
spec_variability = (float)sqrt(spec_variability/NB_FRAMES/NB_TBANDS);
bandwidth_mask = 0;
bandwidth = 0;
maxE = 0;
noise_floor = 5.7e-4f/(1<<(IMAX(0,lsb_depth-8)));
noise_floor *= noise_floor;
below_max_pitch=0;
above_max_pitch=0; for (b=0;b<NB_TBANDS;b++)
{ float E=0; float Em; int band_start, band_end; /* Keep a margin of 300 Hz for aliasing */
band_start = tbands[b];
band_end = tbands[b+1]; for (i=band_start;i<band_end;i++)
{ float binE = out[i].r*(float)out[i].r + out[N-i].r*(float)out[N-i].r
+ out[i].i*(float)out[i].i + out[N-i].i*(float)out[N-i].i;
E += binE;
}
E = SCALE_ENER(E);
maxE = MAX32(maxE, E); if (band_start < 64)
{
below_max_pitch += E;
} else {
above_max_pitch += E;
}
tonal->meanE[b] = MAX32((1-alphaE2)*tonal->meanE[b], E);
Em = MAX32(E, tonal->meanE[b]); /* Consider the band "active" only if all these conditions are met: 1)lessthan90dBbelowthepeakband(maximalmaskingpossibleconsidering boththeATHandtheloudness-dependentslopeofthespreadingfunction) 2)abovethePCMquantizationnoisefloor Weuseb+1becausethefirstCELTbandisn'tincludedintbands[]
*/ if (E*1e9f > maxE && (Em > 3*noise_floor*(band_end-band_start) || E > noise_floor*(band_end-band_start)))
bandwidth = b+1; /* Check if the band is masked (see below). */
is_masked[b] = E < (tonal->prev_bandwidth >= b+1 ? .01f : .05f)*bandwidth_mask; /* Use a simple follower with 13 dB/Bark slope for spreading function. */
bandwidth_mask = MAX32(.05f*bandwidth_mask, E);
} /* Special case for the last two bands, for which we don't have spectrum but only theenergyabove12kHz.Thedifficultyhereisthatthehigh-passweuse leakssomeLFenergy,soweneedtoincreasethethresholdwithoutaccidentallycutting
off the band. */ if (tonal->Fs == 48000) { float noise_ratio; float Em; float E = hp_ener*(1.f/(60*60));
noise_ratio = tonal->prev_bandwidth==20 ? 10.f : 30.f;
#ifdef FIXED_POINT /* silk_resampler_down2_hp() shifted right by an extra 8 bits. */
E *= 256.f*(1.f/Q15ONE)*(1.f/Q15ONE); #endif
above_max_pitch += E;
tonal->meanE[b] = MAX32((1-alphaE2)*tonal->meanE[b], E);
Em = MAX32(E, tonal->meanE[b]); if (Em > 3*noise_ratio*noise_floor*160 || E > noise_ratio*noise_floor*160)
bandwidth = 20; /* Check if the band is masked (see below). */
is_masked[b] = E < (tonal->prev_bandwidth == 20 ? .01f : .05f)*bandwidth_mask;
} if (above_max_pitch > below_max_pitch)
info->max_pitch_ratio = below_max_pitch/above_max_pitch; else
info->max_pitch_ratio = 1; /* In some cases, resampling aliasing can create a small amount of energy in the first band
being cut. So if the last band is masked, we don't include it. */ if (bandwidth == 20 && is_masked[NB_TBANDS])
bandwidth-=2; elseif (bandwidth > 0 && bandwidth <= NB_TBANDS && is_masked[bandwidth-1])
bandwidth--; if (tonal->count<=2)
bandwidth = 20;
frame_loudness = 20*(float)log10(frame_loudness);
tonal->Etracker = MAX32(tonal->Etracker-.003f, frame_loudness);
tonal->lowECount *= (1-alphaE); if (frame_loudness < tonal->Etracker-30)
tonal->lowECount += alphaE;
for (i=0;i<8;i++)
{ float sum=0; for (b=0;b<16;b++)
sum += dct_table[i*16+b]*logE[b];
BFCC[i] = sum;
} for (i=0;i<8;i++)
{ float sum=0; for (b=0;b<16;b++)
sum += dct_table[i*16+b]*.5f*(tonal->highE[b]+tonal->lowE[b]);
midE[i] = sum;
}
void run_analysis(TonalityAnalysisState *analysis, const CELTMode *celt_mode, constvoid *analysis_pcm, int analysis_frame_size, int frame_size, int c1, int c2, int C, opus_int32 Fs, int lsb_depth, downmix_func downmix, AnalysisInfo *analysis_info)
{ int offset; int pcm_len;
analysis_frame_size -= analysis_frame_size&1; if (analysis_pcm != NULL)
{ /* Avoid overflow/wrap-around of the analysis buffer */
analysis_frame_size = IMIN((DETECT_SIZE-5)*Fs/50, analysis_frame_size);
¤ Diese beiden folgenden Angebotsgruppen bietet das Unternehmen0.21Angebot
(Wie Sie bei der Firma Beratungs- und Dienstleistungen beauftragen können 2026-09-27)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.