Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
36 changes: 36 additions & 0 deletions asr.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,36 @@
{
"asr_train_config": "exp/asr_train_asr_conformer_lr2e-3_warmup15k_amp_nondeterministic_raw_en_hugging_face_openai-community-gpt2_sp/config.yaml",
"asr_model_file": "exp/asr_train_asr_conformer_lr2e-3_warmup15k_amp_nondeterministic_raw_en_hugging_face_openai-community-gpt2_sp/valid.acc.ave.pth",
"transducer_conf": null,
"lm_train_config": null,
"lm_file": null,
"ngram_file": null,
"token_type": null,
"bpemodel": null,
"device": "cuda",
"maxlenratio": 0.0,
"minlenratio": 0.0,
"dtype": "float32",
"beam_size": 20,
"ctc_weight": 0.3,
"lm_weight": 0.0,
"ngram_weight": 0.9,
"penalty": 0.0,
"nbest": 1,
"normalize_length": false,
"streaming": false,
"enh_s2t_task": false,
"multi_asr": false,
"quantize_asr_model": false,
"quantize_lm": false,
"quantize_modules": [
"Linear"
],
"quantize_dtype": "qint8",
"hugging_face_decoder": false,
"hugging_face_decoder_conf": {},
"time_sync": false,
"prompt_token_file": null,
"lang_prompt_token": null,
"nlp_prompt_token": null
}
40 changes: 40 additions & 0 deletions lm.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,40 @@
{
"asr_train_config": "exp/asr_train_asr_conformer_lr2e-3_warmup15k_amp_nondeterministic_raw_en_hugging_face_openai-community-gpt2_sp/config.yaml",
"asr_model_file": "exp/asr_train_asr_conformer_lr2e-3_warmup15k_amp_nondeterministic_raw_en_hugging_face_openai-community-gpt2_sp/valid.acc.ave.pth",
"transducer_conf": null,
"lm_train_config": "exp/lm_train_transformer_gpt2_en_hugging_face/config.yaml",
"lm_file": "exp/lm_train_transformer_gpt2_en_hugging_face/valid.loss.ave.pth",
"ngram_file": null,
"token_type": null,
"bpemodel": null,
"device": "cuda",
"maxlenratio": 0.0,
"minlenratio": 0.0,
"dtype": "float32",
"beam_size": 20,
"ctc_weight": 0.3,
"lm_weight": 0.1,
"ngram_weight": 0.9,
"penalty": 0.0,
"nbest": 1,
"normalize_length": false,
"streaming": false,
"enh_s2t_task": false,
"multi_asr": false,
"quantize_asr_model": false,
"quantize_lm": false,
"quantize_modules": [
"Linear"
],
"quantize_dtype": "qint8",
"hugging_face_decoder": false,
"hugging_face_decoder_conf": {},
"time_sync": false,
"prompt_token_file": null,
"lang_prompt_token": null,
"nlp_prompt_token": null,
"partial_ar": false,
"threshold_probability": 0.99,
"max_seq_len": 5,
"max_mask_parallel": -1
}
49 changes: 49 additions & 0 deletions tryModelMerge.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,49 @@
from ..espnet.espnet2.bin.asr_inference import Speech2Text

# (Pdb) p(speech2text_kwargs)
# {'asr_train_config': 'exp/asr_train_asr_conformer_lr2e-3_warmup15k_amp_nondeterministic_raw_en_hugging_face_openai-community-gpt2_sp/config.yaml', 'asr_model_file': 'exp/asr_train_asr_conformer_lr2e-3_warmup15k_amp_nondeterministic_raw_en_hugging_face_openai-community-gpt2_sp/valid.acc.ave.pth', 'transducer_conf': None, 'lm_train_config': None, 'lm_file': None, 'ngram_file': None, 'token_type': None, 'bpemodel': None, 'device': 'cuda', 'maxlenratio': 0.0, 'minlenratio': 0.0, 'dtype': 'float32', 'beam_size': 20, 'ctc_weight': 0.3, 'lm_weight': 0.0, 'ngram_weight': 0.9, 'penalty': 0.0, 'nbest': 1, 'normalize_length': False, 'streaming': False, 'enh_s2t_task': False, 'multi_asr': False, 'quantize_asr_model': False, 'quantize_lm': False, 'quantize_modules': ['Linear'], 'quantize_dtype': 'qint8', 'hugging_face_decoder': False, 'hugging_face_decoder_conf': {}, 'time_sync': False, 'prompt_token_file': None, 'lang_prompt_token': None, 'nlp_prompt_token': None}
speech2text = Speech2Text.from_pretrained(
asr_train_config="/mnt/kiso-qnap/abe/b4/espnet/egs2/librispeech_100/asr1/exp/asr_train_asr_conformer_lr2e-3_warmup15k_amp_nondeterministic_raw_en_hugging_face_openai-community-gpt2_sp/config.yaml",
asr_model_file="/mnt/kiso-qnap/abe/b4/espnet/egs2/librispeech_100/asr1/exp/asr_train_asr_conformer_lr2e-3_warmup15k_amp_nondeterministic_raw_en_hugging_face_openai-community-gpt2_sp/valid.acc.ave.pth",
transducer_conf=None,
lm_train_config=None,
lm_file=None,
ngram_file=None,
token_type=None,
bpemodel=None,
device="cuda",
maxlenratio=0.0,
minlenratio=0.0,
beam_size=20,
ctc_weight=0.3,
lm_weight=0.0,
ngram_weight=0.9,
penalty=0.0,
nbest=1,
)

# (Pdb) p(speech2text_kwargs)
# {'asr_train_config': 'exp/asr_train_asr_conformer_lr2e-3_warmup15k_amp_nondeterministic_raw_en_hugging_face_openai-community-gpt2_sp/config.yaml', 'asr_model_file': 'exp/asr_train_asr_conformer_lr2e-3_warmup15k_amp_nondeterministic_raw_en_hugging_face_openai-community-gpt2_sp/valid.acc.ave.pth', 'transducer_conf': None, 'lm_train_config': 'exp/lm_train_transformer_gpt2_en_hugging_face/config.yaml', 'lm_file': 'exp/lm_train_transformer_gpt2_en_hugging_face/valid.loss.ave.pth', 'ngram_file': None, 'token_type': None, 'bpemodel': None, 'device': 'cuda', 'maxlenratio': 0.0, 'minlenratio': 0.0, 'dtype': 'float32', 'beam_size': 20, 'ctc_weight': 0.3, 'lm_weight': 0.1, 'ngram_weight': 0.9, 'penalty': 0.0, 'nbest': 1, 'normalize_length': False, 'streaming': False, 'enh_s2t_task': False, 'multi_asr': False, 'quantize_asr_model': False, 'quantize_lm': False, 'quantize_modules': ['Linear'], 'quantize_dtype': 'qint8', 'hugging_face_decoder': False, 'hugging_face_decoder_conf': {}, 'time_sync': False, 'prompt_token_file': None, 'lang_prompt_token': None, 'nlp_prompt_token': None, 'partial_ar': False, 'threshold_probability': 0.99, 'max_seq_len': 5, 'max_mask_parallel': -1}
speech2text_for_lm = Speech2Text.from_pretrained(
asr_train_config="/mnt/kiso-qnap2/yuabe/b4/espnet/egs2/librispeech_100/asr1/exp/asr_train_asr_conformer_lr2e-3_warmup15k_amp_nondeterministic_raw_en_hugging_face_openai-community-gpt2_sp/config.yaml",
asr_model_file="/mnt/kiso-qnap2/yuabe/b4/espnet/egs2/librispeech_100/asr1/exp/asr_train_asr_conformer_lr2e-3_warmup15k_amp_nondeterministic_raw_en_hugging_face_openai-community-gpt2_sp/valid.acc.ave.pth",
transducer_conf=None,
lm_train_config="/mnt/kiso-qnap2/yuabe/b4/espnet/egs2/librispeech_100/asr1/exp/lm_train_transformer_gpt2_en_hugging_face/config.yaml",
lm_file="/mnt/kiso-qnap2/yuabe/b4/espnet/egs2/librispeech_100/asr1/exp/lm_train_transformer_gpt2_en_hugging_face/valid.loss.ave.pth",
ngram_file=None,
token_type=None,
bpemodel=None,
device="cuda",
maxlenratio=0.0,
minlenratio=0.0,
beam_size=20,
ctc_weight=0.3,
lm_weight=0.0,
ngram_weight=0.9,
penalty=0.0,
nbest=1,
partial_ar=False,
threshold_probability=0.99,
max_seq_len=5,
max_mask_parallel=-1
)