{ "sample_rate": 16000, "log_prediction": true, "ctc_reduction": "mean_batch", "skip_nan_grad": false, "use_cer": true, "train_ds": { "manifest_filepath": "/kaggle/working/mono_nemo_train.json", "sample_rate": 16000, "batch_size": 4, "shuffle": true, "num_workers": 4, "pin_memory": true, "max_duration": 16.7, "min_duration": 0.1, "is_tarred": false, "tarred_audio_filepaths": null, "shuffle_n": 2048, "bucketing_strategy": "synced_randomized", "bucketing_batch_size": null }, "validation_ds": { "manifest_filepath": "/kaggle/working/mono_nemo_val.json", "sample_rate": 16000, "batch_size": 4, "shuffle": false, "use_start_end_token": false, "num_workers": 4, "pin_memory": true }, "test_ds": { "manifest_filepath": "/kaggle/working/mono_nemo_test.json", "sample_rate": 16000, "batch_size": 4, "shuffle": false, "use_start_end_token": false, "num_workers": 4, "pin_memory": true }, "tokenizer": { "dir": "tokens/tokenizer_spe_unigram_v128_max_200", "type": "bpe", "model_path": "nemo:9a7d38387f7b45a7a6d1c51db1072cbd_tokenizer.model", "vocab_path": "nemo:6729022885274af79d49ebcacbc7b3d3_vocab.txt", "spe_tokenizer_vocab": "nemo:16d5c791dd5049f6a9d102a1cef238df_tokenizer.vocab" }, "preprocessor": { "_target_": "nemo.collections.asr.modules.AudioToMelSpectrogramPreprocessor", "sample_rate": 16000, "normalize": "per_feature", "window_size": 0.025, "window_stride": 0.01, "window": "hann", "features": 80, "n_fft": 512, "log": true, "frame_splicing": 1, "dither": 1e-05, "pad_to": 0, "pad_value": 0.0 }, "spec_augment": { "_target_": "nemo.collections.asr.modules.SpectrogramAugmentation", "freq_masks": 2, "time_masks": 2, "freq_width": 27, "time_width": 70 }, "encoder": { "_target_": "nemo.collections.asr.modules.ConformerEncoder", "feat_in": 80, "feat_out": -1, "n_layers": 18, "d_model": 256, "subsampling": "striding", "subsampling_factor": 4, "subsampling_conv_channels": -1, "causal_downsampling": false, "ff_expansion_factor": 4, "self_attention_model": "rel_pos", "n_heads": 4, "att_context_size": [ -1, -1 ], "att_context_style": "regular", "xscaling": true, "untie_biases": true, "pos_emb_max_len": 5000, "conv_kernel_size": 31, "conv_norm_type": "batch_norm", "conv_context_size": null, "dropout": 0.1, "dropout_pre_encoder": 0.1, "dropout_emb": 0.0, "dropout_att": 0.1, "stochastic_depth_drop_prob": 0.0, "stochastic_depth_mode": "linear", "stochastic_depth_start_layer": 1 }, "decoder": { "_target_": "nemo.collections.asr.modules.ConvASRDecoder", "feat_in": 256, "num_classes": 128, "vocabulary": [ "", "▁", "ा", "ि", "ं", "ल", "ु", "क", "ः", "त", "न", "स", "े", "या", "इ", "र", "्", "▁म", "्व", "गु", "म", "▁व", "्य", "य्", "ाः", "ह", "▁ख", "य", "▁क", "▁स", "▁ब", "▁त", "▁छ", "प", "ू", "▁प", "ख", "ी", "ये", "▁द", "▁थ", "▁न", "चा", "▁ध", "व", "थ", "▁ज", "म्ह", "ना", "ग", "▁अ", "ज", "ँ", "ब", "▁भ", "▁नं", "च", "▁च", "द", "यात", "▁जि", "लि", "ने", "उ", "फ", "ो", "।", "श", "ध", "भ", "छ", "ौ", "झ", "ट", "ै", "॑", "ष", "ण", "”", ",", "घ", "ृ", "ड", "-", "!", "?", "ॉ", ":", "ञ", "ॅ", "‘", "’", "ए", ".", "ङ", "॒", "ओ", "अ", "ठ", "/", "\"", "़", ")", "०", ";", "ऋ", "ढ", "१", "'", "_", "२", "(", "औ", "ई", "‍", "“", "ऱ", "a", "m", "आ", "ऎ", "ऐ", "४", "५", "६", "८", "९", "ḑ" ] }, "interctc": { "loss_weights": [], "apply_at_layers": [] }, "optim": { "name": "adamw", "lr": 0.0001, "betas": [ 0.9, 0.98 ], "weight_decay": 0.001, "sched": { "name": "CosineAnnealing", "min_lr": 1e-06, "warmup_steps": 3000 } }, "target": "nemo.collections.asr.models.ctc_bpe_models.EncDecCTCModelBPE", "nemo_version": "2.1.0", "decoding": { "strategy": "greedy_batch", "preserve_alignments": null, "compute_timestamps": null, "word_seperator": " ", "segment_seperators": [ ".", "!", "?" ], "segment_gap_threshold": null, "ctc_timestamp_type": "all", "batch_dim_index": 0, "greedy": { "preserve_alignments": false, "compute_timestamps": false, "preserve_frame_confidence": false, "confidence_method_cfg": { "name": "entropy", "entropy_type": "tsallis", "alpha": 0.33, "entropy_norm": "exp", "temperature": "DEPRECATED" } }, "beam": { "beam_size": 4, "search_type": "default", "preserve_alignments": false, "compute_timestamps": false, "return_best_hypothesis": true, "beam_alpha": 1.0, "beam_beta": 0.0, "kenlm_path": null, "flashlight_cfg": { "lexicon_path": null, "boost_path": null, "beam_size_token": 16, "beam_threshold": 20.0, "unk_weight": -Infinity, "sil_weight": 0.0 }, "pyctcdecode_cfg": { "beam_prune_logp": -10.0, "token_min_logp": -5.0, "prune_history": false, "hotwords": null, "hotword_weight": 10.0 } }, "wfst": { "beam_size": 4, "search_type": "riva", "return_best_hypothesis": true, "preserve_alignments": false, "compute_timestamps": false, "decoding_mode": "nbest", "open_vocabulary_decoding": false, "beam_width": 10.0, "lm_weight": 1.0, "device": "cuda", "arpa_lm_path": null, "wfst_lm_path": null, "riva_decoding_cfg": {}, "k2_decoding_cfg": { "search_beam": 20.0, "output_beam": 10.0, "min_active_states": 30, "max_active_states": 10000 } }, "confidence_cfg": { "preserve_frame_confidence": false, "preserve_token_confidence": false, "preserve_word_confidence": false, "exclude_blank": true, "aggregation": "min", "tdt_include_duration": false, "method_cfg": { "name": "entropy", "entropy_type": "tsallis", "alpha": 0.33, "entropy_norm": "exp", "temperature": "DEPRECATED" } }, "temperature": 1.0 } }