You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
53 lines
1.7 KiB
53 lines
1.7 KiB
###########################################
|
|
# Data #
|
|
###########################################
|
|
# we should explicitly specify the wav path of vox2 audio data converted from m4a
|
|
vox2_base_path:
|
|
augment: True
|
|
batch_size: 16
|
|
num_workers: 2
|
|
num_speakers: 7205 # 1211 vox1, 5994 vox2, 7205 vox1+2, test speakers: 41
|
|
shuffle: True
|
|
random_chunk: True
|
|
|
|
###########################################################
|
|
# FEATURE EXTRACTION SETTING #
|
|
###########################################################
|
|
# currently, we only support fbank
|
|
sr: 16000 # sample rate
|
|
n_mels: 80
|
|
window_size: 400 #25ms, sample rate 16000, 25 * 16000 / 1000 = 400
|
|
hop_size: 160 #10ms, sample rate 16000, 10 * 16000 / 1000 = 160
|
|
|
|
###########################################################
|
|
# MODEL SETTING #
|
|
###########################################################
|
|
# currently, we only support ecapa-tdnn in the ecapa_tdnn.yaml
|
|
# if we want use another model, please choose another configuration yaml file
|
|
model:
|
|
input_size: 80
|
|
# "channels": [512, 512, 512, 512, 1536],
|
|
channels: [1024, 1024, 1024, 1024, 3072]
|
|
kernel_sizes: [5, 3, 3, 3, 1]
|
|
dilations: [1, 2, 3, 4, 1]
|
|
attention_channels: 128
|
|
lin_neurons: 192
|
|
|
|
###########################################
|
|
# Training #
|
|
###########################################
|
|
seed: 1986 # according from speechbrain configuration
|
|
epochs: 10
|
|
save_interval: 1
|
|
log_interval: 1
|
|
learning_rate: 1e-8
|
|
|
|
|
|
###########################################
|
|
# Testing #
|
|
###########################################
|
|
global_embedding_norm: True
|
|
embedding_mean_norm: True
|
|
embedding_std_norm: False
|
|
|