-
Notifications
You must be signed in to change notification settings - Fork 5.4k
Augmentation recipe for swbd #1112
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from 8 commits
92ad8ba
0bcf41e
178c9d1
dc13729
82cd6f7
a4ee796
67673fa
823bcac
01e47f6
b8453c0
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -1 +1 @@ | ||
| tuning/run_tdnn_7e.sh | ||
| tuning/run_tdnn_7g.sh | ||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more.
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. I am asking this question as we will not be able to compare our results with other papers. We don't already do it anyway as we use speed-perturbation. So @tomkocse could you just add a commented line in this script #for swbd recipe without the reverberation of training data use the following script
# it is similar to run_tdnn_7g.sh except for the run_ivector_common.sh being called.
# tuning/run_tdnn_7f.sh |
||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,211 @@ | ||
| #!/bin/bash | ||
|
|
||
| # 7f is as 7e, but adding the max-change-per-component to the neural net training | ||
| # which affects results slightly | ||
| # local/chain/compare_wer.sh 7e 7f | ||
| # System 7e 7f | ||
| # WER on train_dev(tg) 14.41 14.46 | ||
| # WER on train_dev(fg) 13.39 13.23 | ||
| # WER on eval2000(tg) 16.9 17.0 | ||
| # WER on eval2000(fg) 15.3 15.4 | ||
| # Final train prob -0.0853629 -0.0882071 | ||
| # Final valid prob -0.110972 -0.107545 | ||
| # Final train prob (xent) -1.25237 -1.26246 | ||
| # Final valid prob (xent) -1.36715 -1.35525 | ||
|
|
||
|
|
||
| set -e | ||
|
|
||
| # configs for 'chain' | ||
| affix= | ||
| stage=12 | ||
| train_stage=-10 | ||
| get_egs_stage=-10 | ||
| speed_perturb=true | ||
| dir=exp/chain/tdnn_7f # Note: _sp will get added to this if $speed_perturb == true. | ||
| decode_iter= | ||
|
|
||
| # TDNN options | ||
| # this script uses the new tdnn config generator so it needs a final 0 to reflect that the final layer input has no splicing | ||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. This comment makes sense when it is next to the splice_indexes specification. |
||
| splice_indexes="-1,0,1 -1,0,1 -1,0,1 -3,0,3 -3,0,3 -6,0,6 0" | ||
| # smoothing options | ||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. what smoothing are you referring to ? |
||
| self_repair_scale=0.00001 | ||
| # training options | ||
| num_epochs=4 | ||
| initial_effective_lrate=0.001 | ||
| final_effective_lrate=0.0001 | ||
| leftmost_questions_truncate=-1 | ||
| max_param_change=2.0 | ||
| final_layer_normalize_target=0.5 | ||
| num_jobs_initial=3 | ||
| num_jobs_final=16 | ||
| minibatch_size=128 | ||
| relu_dim=625 | ||
| frames_per_eg=150 | ||
| remove_egs=false | ||
| common_egs_dir= | ||
| xent_regularize=0.1 | ||
|
|
||
| # End configuration section. | ||
| echo "$0 $@" # Print the command line for logging | ||
|
|
||
| . ./cmd.sh | ||
| . ./path.sh | ||
| . ./utils/parse_options.sh | ||
|
|
||
| if ! cuda-compiled; then | ||
| cat <<EOF && exit 1 | ||
| This script is intended to be used with GPUs but you have not compiled Kaldi with CUDA | ||
| If you want to use GPUs (and have them), go to src/, and configure and make on a machine | ||
| where "nvcc" is installed. | ||
| EOF | ||
| fi | ||
|
|
||
| # The iVector-extraction and feature-dumping parts are the same as the standard | ||
| # nnet3 setup, and you can skip them by setting "--stage 8" if you have already | ||
| # run those things. | ||
|
|
||
| suffix= | ||
| if [ "$speed_perturb" == "true" ]; then | ||
| suffix=_sp | ||
| fi | ||
|
|
||
| dir=${dir}${affix:+_$affix}$suffix | ||
| train_set=train_nodup$suffix | ||
| ali_dir=exp/tri4_ali_nodup$suffix | ||
| treedir=exp/chain/tri5_7d_tree$suffix | ||
| lang=data/lang_chain_2y | ||
|
|
||
|
|
||
| # if we are using the speed-perturbed data we need to generate | ||
| # alignments for it. | ||
| local/nnet3/run_ivector_common.sh --stage $stage \ | ||
| --speed-perturb $speed_perturb \ | ||
| --generate-alignments $speed_perturb || exit 1; | ||
|
|
||
|
|
||
| if [ $stage -le 9 ]; then | ||
| # Get the alignments as lattices (gives the LF-MMI training more freedom). | ||
| # use the same num-jobs as the alignments | ||
| nj=$(cat exp/tri4_ali_nodup$suffix/num_jobs) || exit 1; | ||
| steps/align_fmllr_lats.sh --nj $nj --cmd "$train_cmd" data/$train_set \ | ||
| data/lang exp/tri4 exp/tri4_lats_nodup$suffix | ||
| rm exp/tri4_lats_nodup$suffix/fsts.*.gz # save space | ||
| fi | ||
|
|
||
|
|
||
| if [ $stage -le 10 ]; then | ||
| # Create a version of the lang/ directory that has one state per phone in the | ||
| # topo file. [note, it really has two states.. the first one is only repeated | ||
| # once, the second one has zero or more repeats.] | ||
| rm -rf $lang | ||
| cp -r data/lang $lang | ||
| silphonelist=$(cat $lang/phones/silence.csl) || exit 1; | ||
| nonsilphonelist=$(cat $lang/phones/nonsilence.csl) || exit 1; | ||
| # Use our special topology... note that later on may have to tune this | ||
| # topology. | ||
| steps/nnet3/chain/gen_topo.py $nonsilphonelist $silphonelist >$lang/topo | ||
| fi | ||
|
|
||
| if [ $stage -le 11 ]; then | ||
| # Build a tree using our new topology. This is the critically different | ||
| # step compared with other recipes. | ||
| steps/nnet3/chain/build_tree.sh --frame-subsampling-factor 3 \ | ||
| --leftmost-questions-truncate $leftmost_questions_truncate \ | ||
| --context-opts "--context-width=2 --central-position=1" \ | ||
| --cmd "$train_cmd" 7000 data/$train_set $lang $ali_dir $treedir | ||
| fi | ||
|
|
||
| if [ $stage -le 12 ]; then | ||
| echo "$0: creating neural net configs"; | ||
| if [ ! -z "$relu_dim" ]; then | ||
| dim_opts="--relu-dim $relu_dim" | ||
| else | ||
| dim_opts="--pnorm-input-dim $pnorm_input_dim --pnorm-output-dim $pnorm_output_dim" | ||
| fi | ||
|
|
||
| # create the config files for nnet initialization | ||
| repair_opts=${self_repair_scale:+" --self-repair-scale-nonlinearity $self_repair_scale "} | ||
|
|
||
| steps/nnet3/tdnn/make_configs.py \ | ||
| $repair_opts \ | ||
| --feat-dir data/${train_set}_hires \ | ||
| --ivector-dir exp/nnet3/ivectors_${train_set} \ | ||
| --tree-dir $treedir \ | ||
| $dim_opts \ | ||
| --splice-indexes "$splice_indexes" \ | ||
| --use-presoftmax-prior-scale false \ | ||
| --xent-regularize $xent_regularize \ | ||
| --xent-separate-forward-affine true \ | ||
| --include-log-softmax false \ | ||
| --final-layer-normalize-target $final_layer_normalize_target \ | ||
| $dir/configs || exit 1; | ||
| fi | ||
|
|
||
|
|
||
|
|
||
| if [ $stage -le 13 ]; then | ||
| if [[ $(hostname -f) == *.clsp.jhu.edu ]] && [ ! -d $dir/egs/storage ]; then | ||
| utils/create_split_dir.pl \ | ||
| /export/b0{5,6,7,8}/$USER/kaldi-data/egs/swbd-$(date +'%m_%d_%H_%M')/s5c/$dir/egs/storage $dir/egs/storage | ||
| fi | ||
|
|
||
| steps/nnet3/chain/train.py --stage $train_stage \ | ||
| --cmd "$decode_cmd" \ | ||
| --feat.online-ivector-dir exp/nnet3/ivectors_${train_set} \ | ||
| --feat.cmvn-opts "--norm-means=false --norm-vars=false" \ | ||
| --chain.xent-regularize $xent_regularize \ | ||
| --chain.leaky-hmm-coefficient 0.1 \ | ||
| --chain.l2-regularize 0.00005 \ | ||
| --chain.apply-deriv-weights false \ | ||
| --chain.lm-opts="--num-extra-lm-states=2000" \ | ||
| --egs.dir "$common_egs_dir" \ | ||
| --egs.stage $get_egs_stage \ | ||
| --egs.opts "--frames-overlap-per-eg 0" \ | ||
| --egs.chunk-width $frames_per_eg \ | ||
| --trainer.num-chunk-per-minibatch $minibatch_size \ | ||
| --trainer.frames-per-iter 1500000 \ | ||
| --trainer.num-epochs $num_epochs \ | ||
| --trainer.optimization.num-jobs-initial $num_jobs_initial \ | ||
| --trainer.optimization.num-jobs-final $num_jobs_final \ | ||
| --trainer.optimization.initial-effective-lrate $initial_effective_lrate \ | ||
| --trainer.optimization.final-effective-lrate $final_effective_lrate \ | ||
| --trainer.max-param-change $max_param_change \ | ||
| --cleanup.remove-egs $remove_egs \ | ||
| --feat-dir data/${train_set}_hires \ | ||
| --tree-dir $treedir \ | ||
| --lat-dir exp/tri4_lats_nodup$suffix \ | ||
| --dir $dir || exit 1; | ||
|
|
||
| fi | ||
|
|
||
| if [ $stage -le 14 ]; then | ||
| # Note: it might appear that this $lang directory is mismatched, and it is as | ||
| # far as the 'topo' is concerned, but this script doesn't read the 'topo' from | ||
| # the lang directory. | ||
| utils/mkgraph.sh --left-biphone --self-loop-scale 1.0 data/lang_sw1_tg $dir $dir/graph_sw1_tg | ||
| fi | ||
|
|
||
| decode_suff=sw1_tg | ||
| graph_dir=$dir/graph_sw1_tg | ||
| if [ $stage -le 15 ]; then | ||
| iter_opts= | ||
| if [ ! -z $decode_iter ]; then | ||
| iter_opts=" --iter $decode_iter " | ||
| fi | ||
| for decode_set in train_dev eval2000; do | ||
| ( | ||
| steps/nnet3/decode.sh --acwt 1.0 --post-decode-acwt 10.0 \ | ||
| --nj 50 --cmd "$decode_cmd" $iter_opts \ | ||
| --online-ivector-dir exp/nnet3/ivectors_${decode_set} \ | ||
| $graph_dir data/${decode_set}_hires $dir/decode_${decode_set}${decode_iter:+_$decode_iter}_${decode_suff} || exit 1; | ||
| if $has_fisher; then | ||
| steps/lmrescore_const_arpa.sh --cmd "$decode_cmd" \ | ||
| data/lang_sw1_{tg,fsh_fg} data/${decode_set}_hires \ | ||
| $dir/decode_${decode_set}${decode_iter:+_$decode_iter}_sw1_{tg,fsh_fg} || exit 1; | ||
| fi | ||
| ) & | ||
| done | ||
| fi | ||
| wait; | ||
| exit 0; | ||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
Does the model converge after 2 epochs of training ? Could you please post the log-likelihood plots here.