|
|
|
@ -1,5 +1,6 @@
|
|
|
|
# network architecture
|
|
|
|
############################################
|
|
|
|
model:
|
|
|
|
# Network Architecture #
|
|
|
|
|
|
|
|
############################################
|
|
|
|
cmvn_file:
|
|
|
|
cmvn_file:
|
|
|
|
cmvn_file_type: "json"
|
|
|
|
cmvn_file_type: "json"
|
|
|
|
# encoder related
|
|
|
|
# encoder related
|
|
|
|
@ -23,7 +24,6 @@ model:
|
|
|
|
use_dynamic_chunk: true
|
|
|
|
use_dynamic_chunk: true
|
|
|
|
cnn_module_norm: 'layer_norm' # using nn.LayerNorm makes model converge faster
|
|
|
|
cnn_module_norm: 'layer_norm' # using nn.LayerNorm makes model converge faster
|
|
|
|
use_dynamic_left_chunk: false
|
|
|
|
use_dynamic_left_chunk: false
|
|
|
|
|
|
|
|
|
|
|
|
# decoder related
|
|
|
|
# decoder related
|
|
|
|
decoder: transformer
|
|
|
|
decoder: transformer
|
|
|
|
decoder_conf:
|
|
|
|
decoder_conf:
|
|
|
|
@ -34,21 +34,25 @@ model:
|
|
|
|
positional_dropout_rate: 0.1
|
|
|
|
positional_dropout_rate: 0.1
|
|
|
|
self_attention_dropout_rate: 0.0
|
|
|
|
self_attention_dropout_rate: 0.0
|
|
|
|
src_attention_dropout_rate: 0.0
|
|
|
|
src_attention_dropout_rate: 0.0
|
|
|
|
|
|
|
|
|
|
|
|
# hybrid CTC/attention
|
|
|
|
# hybrid CTC/attention
|
|
|
|
model_conf:
|
|
|
|
model_conf:
|
|
|
|
ctc_weight: 0.3
|
|
|
|
ctc_weight: 0.3
|
|
|
|
lsm_weight: 0.1 # label smoothing option
|
|
|
|
lsm_weight: 0.1 # label smoothing option
|
|
|
|
length_normalized_loss: false
|
|
|
|
length_normalized_loss: false
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
###########################################
|
|
|
|
|
|
|
|
# Data #
|
|
|
|
|
|
|
|
###########################################
|
|
|
|
|
|
|
|
|
|
|
|
data:
|
|
|
|
|
|
|
|
train_manifest: data/manifest.train
|
|
|
|
train_manifest: data/manifest.train
|
|
|
|
dev_manifest: data/manifest.dev
|
|
|
|
dev_manifest: data/manifest.dev
|
|
|
|
test_manifest: data/manifest.test
|
|
|
|
test_manifest: data/manifest.test
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
collator:
|
|
|
|
###########################################
|
|
|
|
|
|
|
|
# Dataloader #
|
|
|
|
|
|
|
|
###########################################
|
|
|
|
|
|
|
|
|
|
|
|
vocab_filepath: data/lang_char/vocab.txt
|
|
|
|
vocab_filepath: data/lang_char/vocab.txt
|
|
|
|
unit_type: 'char'
|
|
|
|
unit_type: 'char'
|
|
|
|
augmentation_config: conf/preprocess.yaml
|
|
|
|
augmentation_config: conf/preprocess.yaml
|
|
|
|
@ -69,8 +73,9 @@ collator:
|
|
|
|
subsampling_factor: 1
|
|
|
|
subsampling_factor: 1
|
|
|
|
num_encs: 1
|
|
|
|
num_encs: 1
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
###########################################
|
|
|
|
training:
|
|
|
|
# training #
|
|
|
|
|
|
|
|
###########################################
|
|
|
|
n_epoch: 240
|
|
|
|
n_epoch: 240
|
|
|
|
accum_grad: 2
|
|
|
|
accum_grad: 2
|
|
|
|
global_grad_clip: 5.0
|
|
|
|
global_grad_clip: 5.0
|
|
|
|
@ -87,17 +92,3 @@ training:
|
|
|
|
kbest_n: 50
|
|
|
|
kbest_n: 50
|
|
|
|
latest_n: 5
|
|
|
|
latest_n: 5
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
decoding:
|
|
|
|
|
|
|
|
beam_size: 10
|
|
|
|
|
|
|
|
batch_size: 128
|
|
|
|
|
|
|
|
error_rate_type: cer
|
|
|
|
|
|
|
|
decoding_method: attention # 'attention', 'ctc_greedy_search', 'ctc_prefix_beam_search', 'attention_rescoring'
|
|
|
|
|
|
|
|
ctc_weight: 0.5 # ctc weight for attention rescoring decode mode.
|
|
|
|
|
|
|
|
decoding_chunk_size: -1 # decoding chunk size. Defaults to -1.
|
|
|
|
|
|
|
|
# <0: for decoding, use full chunk.
|
|
|
|
|
|
|
|
# >0: for decoding, use fixed chunk size as set.
|
|
|
|
|
|
|
|
# 0: used for training, it's prohibited here.
|
|
|
|
|
|
|
|
num_decoding_left_chunks: -1 # number of left chunks for decoding. Defaults to -1.
|
|
|
|
|
|
|
|
simulate_streaming: False # simulate streaming inference. Defaults to False.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|