s3prl
Self-Supervised Speech Pre-training and Representation Learning Toolkit
File Explorer
Download Latest Version (.zip)- bug_report.md
- feature_request.md
- ci.yml
- doc.yml
- espnet.yml
- format.py
- custom.css
- custom.js
- custom-module-template.rst
- general.rst
- private.rst
- public.rst
- upstream.rst
- installation.rst
- problem.rst
- upstream_collection.rst
- conf.py
- index.rst
- is_valid.py
- from_scratch_tutorial.md
- make.bat
- Makefile
- README.md
- rebuild_docs.sh
- pretrain.py
- train.py
- inference.py
- train.py
- train_with_lightning.py
- inference.py
- train.py
- train_with_lightning.py
- inference.py
- train.py
- train_with_lightning.py
- customize.py
- run_asr.sh
- run_sid.sh
- install_espnet.sh
- license.svg
- S3PRL-integration.png
- S3PRL-interface.png
- S3PRL-logo.png
- all.txt
- dev.txt
- install.txt
- __init__.py
- base.py
- fluent_speech_commands.py
- iemocap.py
- librilight.py
- librispeech.py
- quesst14.py
- snips.py
- speech_commands.py
- voxceleb1sid.py
- voxceleb1sv.py
- __init__.py
- base.py
- diarization.py
- encode.py
- frame_label.py
- load_audio.py
- util.py
- __init__.py
- category.py
- g2p.py
- tokenizer.py
- vocabulary.py
- __init__.py
- balanced_weighted_sampler.py
- distributed_sampler.py
- fixed_batch_size_batch_sampler.py
- group_same_item_sampler.py
- max_timestamp_batch_sampler.py
- sorted_sampler.py
- __init__.py
- collate_fn.py
- __init__.py
- autoregressive_prediction_pipes.py
- base.py
- chunking.py
- common_pipes.py
- extract_feat_pipes.py
- hear_timestamp.py
- masked_reconstruction_pipes.py
- multiclass_tagging.py
- noise_augmentation_pipes.py
- norm_wav_pipes.py
- pretrain_apc_pipe.py
- pretrain_audioalbert_pipe.py
- pretrain_mockingjay_pipe.py
- pretrain_npc_pipe.py
- pretrain_tera_pipe.py
- speaker_verification_pipe.py
- speech2phoneme_pipe.py
- speech2text_pipe.py
- utterance_classification_pipe.py
- valid_label_mask_pipes.py
- dev_list.txt
- test_list.txt
- train_list.txt
- vcc2020_download.sh
- vctk_download.sh
- batch_vc_decode.sh
- config_ar_taco2.yaml
- custom_decode.sh
- dataset.py
- decode.sh
- evaluate.py
- expert.py
- find_best_epoch.py
- model.py
- README.md
- requirements.txt
- utils.py
- vc_evaluate.py
- vc_train.sh
- vocoder_download.sh
- E_dev_list.txt
- E_train_list.txt
- eval_list.txt
- F_dev_list.txt
- F_train_list.txt
- G_dev_list.txt
- G_train_list.txt
- M_dev_list.txt
- M_train_list.txt
- ref_list.txt
- data_download.sh
- f0.yaml
- thresholds.yaml
- __init__.py
- batch_vc_decode.sh
- batch_vc_train.sh
- config.yaml
- config_simple.yaml
- config_simple_ar.yaml
- config_taco2_ar.yaml
- custom_decode.sh
- dataset.py
- decode.sh
- evaluate.py
- expert.py
- find_best_epoch.py
- model.py
- README.md
- requirements.txt
- utils.py
- vc_evaluate.py
- vocoder_download.sh
- __init__.py
- char.dict
- config.yaml
- dataset.py
- dictionary.py
- expert.py
- fairseq_dictionary.py
- model.py
- w2l_decoder.py
- __init__.py
- __init__.py
- config.yaml
- dataset.py
- expert.py
- model.py
- __init__.py
- config.yaml
- dataset.py
- expert.py
- model.py
- common_voice.py
- common_voice_preprocess.py
- downsample_cv.py
- libriphone.py
- librispeech.py
- preprocess_cv.sh
- snips.py
- cv_ar.yaml
- cv_es.yaml
- cv_zh.yaml
- ar_char.txt
- en_char.txt
- es_char.txt
- zh-CN_char.txt
- librispeech-lexicon-200k-g2p.txt
- librispeech-lexicon-allothers-g2p.txt
- librispeech-lexicon.txt
- character.txt
- phoneme.txt
- __init__.py
- data.py
- expert.py
- libriphone.yaml
- librispeech.yaml
- metric.py
- README.md
- sbcsae.yaml
- snips.yaml
- text.py
- __init__.py
- config.yaml
- dataset.py
- expert.py
- make_rttm.py
- model.py
- report.sh
- score.sh
- utils.py
- mockingjay_tera.md
- more_tasks.md
- superb.md
- superb_artifacts.md
- test_meta_data.json
- train_meta_data.json
- test_meta_data.json
- train_meta_data.json
- test_meta_data.json
- train_meta_data.json
- test_meta_data.json
- train_meta_data.json
- test_meta_data.json
- train_meta_data.json
- __init__.py
- config.yaml
- dataset.py
- expert.py
- IEMOCAP_preprocess.py
- model.py
- cfg_voicebank.yaml
- data_prepare.py
- data_prepare.py
- __init__.py
- dataset.py
- expert.py
- loss.py
- model.py
- cfg_voicebank.yaml
- data_prepare.py
- data_prepare.py
- __init__.py
- dataset.py
- expert.py
- loss.py
- model.py
- README.md
- __init__.py
- config.yaml
- dataset.py
- expert.py
- model.py
- __init__.py
- config.yaml
- dataset.py
- expert.py
- model.py
- __init__.py
- __init__.py
- system_mos.csv
- test_judge.csv
- train_judge.csv
- valid_judge.csv
- apc_idtable.pkl
- tera_idtable.pkl
- wav2vec2_idtable.pkl
- __init__.py
- config.yaml
- dataset.py
- expert.py
- model.py
- README.md
- CMU_MOSEI_Labels.csv
- convert_label.py
- convert_label.sh
- README.md
- segment_audio.py
- __init__.py
- config.yaml
- dataset.py
- expert.py
- model.py
- README.md
- __init__.py
- config.yaml
- expert.py
- model.py
- converted_aligned_phones.txt
- test_split.txt
- train_split.txt
- __init__.py
- config.yaml
- dataset.py
- expert.py
- model.py
- __init__.py
- config.yaml
- expert.py
- model.py
- __init__.py
- config.yaml
- dataset.py
- expert.py
- __init__.py
- config.yaml
- expert.py
- model.py
- quesst14_testset.py
- quesst14_trainset.py
- cfg.yaml
- data_prepare.py
- subsample.py
- __init__.py
- dataset.py
- expert.py
- loss.py
- model.py
- cfg.yaml
- data_prepare.py
- subsample.py
- __init__.py
- dataset.py
- expert.py
- loss.py
- model.py
- README.md
- __init__.py
- __init__.py
- config.yaml
- expert.py
- test_split.txt
- train_split.txt
- __init__.py
- config.yaml
- dataset.py
- expert.py
- model.py
- __init__.py
- config.yaml
- dataset.py
- expert.py
- model.py
- prepare_clean_paired_corpus.py
- prepare_covo.sh
- prepare_create_config.py
- prepare_data.py
- prepare_data.sh
- prepare_gen_fairseq_vocab.py
- prepare_normalize_tsv.py
- __init__.py
- AdditionalDataset.py
- config.yaml
- count_sacreBLEU.py
- expert.py
- Fairseq_SpeechToTextDataset.py
- README.md
- S3prl_SpeechToTextTask.py
- __init__.py
- __init__.py
- placeholder_to_create_folder
- amsoftmax.yaml
- amsoftmax_attentive.yaml
- attentive.yaml
- vanilla.yaml
- dev_meta_data.txt
- dev_meta_data_voxceleb2.txt
- dev_meta_speaker_ids.txt
- dev_speaker_ids.txt
- __init__.py
- config.yaml
- dataset.py
- expert.py
- model.py
- preprocess.py
- report.sh
- test_expdir.sh
- utils.py
- voxceleb1_test_v2.txt
- __init__.py
- config.yaml
- expert.py
- model.py
- quesst14_dataset.py
- sws2013_dataset.py
- sws2013_testset.py
- __init__.py
- config.yaml
- expert.py
- hubconf.py
- model.py
- upstream_expert.py
- __init__.py
- config.yaml
- expert.py
- model.py
- converted_aligned_phones.txt
- test_split.txt
- train_split.txt
- __init__.py
- config.yaml
- dataset.py
- expert.py
- model.py
- __init__.py
- config.yaml
- expert.py
- model.py
- __init__.py
- config.yaml
- dataset.py
- expert.py
- model.py
- veri_test_class.txt
- __init__.py
- config.yaml
- expert.py
- cache_dev_segment.p
- cache_test_segment.p
- cache_Voxceleb1.p
- cache_Voxceleb2.p
- placeholder_to_create_folder
- dev_meta_data.txt
- dev_speaker_ids.txt
- __init__.py
- config.yaml
- dataset.py
- expert.py
- model.py
- utils.py
- dev_meta_data.txt
- dev_speaker_ids.txt
- __init__.py
- config.yaml
- dataset.py
- expert.py
- model.py
- preprocess.py
- utils.py
- __init__.py
- model.py
- README.md
- runner.py
- specaug.py
- __init__.py
- common.py
- diarization.py
- slot_filling.py
- __init__.py
- beam_decoder.py
- cnn_npc.py
- common.py
- hear.py
- interface.py
- linear.py
- pit.py
- pooling.py
- predictor_identity.py
- predictor_mockingjay.py
- rnn.py
- rnn_apc.py
- speaker_loss.py
- speaker_model.py
- specaug.py
- transformer_mockingjay.py
- upstream.py
- vq_apc.py
- extract_mosei.py
- length_mosei.py
- segment_mosei.py
- ark2libri.py
- ark2timit.py
- ark2voxceleb.py
- generate_len_for_bucket.py
- get_libri_words_not_in_lexicon.py
- preprocess_alignment.py
- preprocess_any.py
- preprocess_libri.py
- preprocess_mosi.py
- preprocess_timit.py
- snips_prepare_data.sh
- snips_preprocess.py
- snips_text_norm.py
- split_long_utter_to_short.py
- timit2ark.py
- __init__.py
- config_model.yaml
- config_runner.yaml
- dataset.py
- pretrain_expert.py
- __init__.py
- config_model.yaml
- config_runner.yaml
- pretrain_expert.py
- config_model.yaml
- config_runner.yaml
- dataset.py
- pretrain_expert.py
- __init__.py
- config_model.yaml
- config_model_large.yaml
- config_runner.yaml
- dataset.py
- pretrain_expert.py
- task.py
- __init__.py
- config_model.yaml
- config_runner.yaml
- dataset.py
- pretrain_expert.py
- __init__.py
- config_model.yaml
- config_runner.yaml
- dataset.py
- pretrain_expert.py
- task.py
- fbankBase-T-F.yaml
- logMelBase-F-M.yaml
- logMelBase-F.yaml
- logMelBase-M.yaml
- logMelBase-T-F-M.yaml
- logMelBase-T-F.yaml
- logMelBase-T-M.yaml
- logMelBase-T.yaml
- logMelSlim-T-F-M.yaml
- __init__.py
- config_model.yaml
- config_runner.yaml
- config_runner_v2.yaml
- pretrain_expert.py
- __init__.py
- config_model.yaml
- config_runner.yaml
- pretrain_expert.py
- __init__.py
- bucket_dataset.py
- README.md
- runner.py
- __init__.py
- run.py
- superb_asr.py
- superb_pr.py
- superb_sf.py
- __init__.py
- run.py
- superb_asv.py
- __init__.py
- _hear_util.py
- example.py
- hear_beijing_opera.py
- hear_cremad.py
- hear_dcase_2016_task2.py
- hear_esc50.py
- hear_fsd.py
- hear_gsc5hr.py
- hear_gtzan.py
- hear_gtzan_music_speech.py
- hear_gunshot.py
- hear_libricount.py
- hear_maestro.py
- hear_nsynth5hr.py
- hear_stroke.py
- hear_tonic.py
- hear_vocal.py
- hear_vox_lingual.py
- run.py
- superb_er.py
- superb_ic.py
- superb_ks.py
- superb_sid.py
- __init__.py
- run.py
- superb_sd.py
- util.py
- beijing_opera.py
- crema_d.py
- dcase_2016_task2.py
- esc50.py
- fsd.py
- gsc5hr.py
- gtzan.py
- gtzan_music_speech.py
- gunshot.py
- libricount.py
- maestro.py
- nsynth5hr.py
- scene.py
- stroke.py
- timestamp.py
- tonic.py
- vocal.py
- vox_lingua.py
- apc.py
- audioalbert.py
- base.py
- mockingjay.py
- npc.py
- tera.py
- vqapc.py
- asr.py
- base.py
- er.py
- ic.py
- ks.py
- pr.py
- qbe.py
- sd.py
- sf.py
- sid.py
- sv.py
- __init__.py
- base.py
- demo_submit.sh
- submit.py
- __init__.py
- _hear_score.py
- autoregressive_reconstruction_task.py
- base.py
- diarization.py
- dump_feature.py
- event_prediction.py
- feat_reconstruction_task.py
- scene_prediction.py
- speaker_verification_task.py
- speech2text_ctc_task.py
- utterance_classification_task.py
- __init__.py
- apc.py
- audio.py
- expert.py
- hubconf.py
- vq.py
- __init__.py
- ast_models.py
- expert.py
- hubconf.py
- __init__.py
- builder.py
- expert.py
- hubconf.py
- __init__.py
- expert.py
- extracter.py
- fbank.yaml
- fbank_no_cmvn.yaml
- hubconf.py
- linear.yaml
- mel.yaml
- mfcc.yaml
- preprocessor.py
- spectrogram.yaml
- __init__.py
- byol_a.py
- config.yaml
- expert.py
- hubconf.py
- __init__.py
- audio_ntt.py
- clstm.py
- cvt.py
- resnetish.py
- sst.py
- utils.py
- __init__.py
- augmentations.py
- byol_pytorch.py
- common.py
- dataset.py
- __init__.py
- config.yaml
- serab.py
- utils.py
- __init__.py
- expert.py
- hubconf.py
- __init__.py
- cpc_default_config.py
- expert.py
- feature_loader.py
- hubconf.py
- model.py
- __init__.py
- convert.py
- data2vec_model.py
- expert.py
- hubconf.py
- __init__.py
- audio.py
- decoar.py
- expert.py
- hubconf.py
- __init__.py
- audio.py
- decoar2.py
- expert.py
- hubconf.py
- __init__.py
- audio.py
- decoar.py
- expert.py
- hubconf.py
- __init__.py
- builder.py
- expert.py
- hubconf.py
- model.py
- module.py
- README.md
- __init__.py
- expert.py
- hubconf.py
- __init__.py
- expert.py
- hubconf.py
- README.md
- __init__.py
- expert.py
- hubconf.py
- __init__.py
- expert.py
- hubconf.py
- __init__.py
- convert.py
- expert.py
- hubconf.py
- hubert_model.py
- __init__.py
- fairseq_utils.py
- sliding_attn.py
- __init__.py
- fairseq_modules.py
- scaling_conv.py
- scaling_layernorm.py
- scaling_linear.py
- scaling_multihead.py
- scaling_transformer.py
- w2v2_modules.py
- __init__.py
- lighthubert.py
- __init__.py
- expert.py
- hubconf.py
- __init__.py
- expert.py
- hubconf.py
- log_stft_mag.yaml
- stft_mag.yaml
- __init__.py
- expert.py
- hubconf.py
- load_model.py
- mae_ast.py
- __init__.py
- builder.py
- expert.py
- hubconf.py
- model.py
- options.yaml
- __init__.py
- expert.py
- hubconf.py
- model.py
- README.md
- utility.py
- __init__.py
- convert.py
- expert.py
- hubconf.py
- hubert_model.py
- __init__.py
- audio.py
- expert.py
- hubconf.py
- npc.py
- vq.py
- __init__.py
- expert.py
- hubconf.py
- README.md
- requirements.txt
- __init__.py
- vit_helpers.py
- __init__.py
- passt.py
- preprocess.py
- __init__.py
- api.py
- base.py
- base20sec.py
- base2level.py
- base2levelmel.py
- base30sec.py
- hop100base.py
- hop100base2lvl.py
- hop100base2lvlmel.py
- hop160base.py
- hop160base2lvl.py
- hop160base2lvlmel.py
- openmic2008.py
- wrapper.py
- __init__.py
- expert.py
- hubconf.py
- __init__.py
- convert.py
- dictionary.py
- expert.py
- hubconf.py
- roberta_model.py
- __init__.py
- builder.py
- expert.py
- hubconf.py
- __init__.py
- ast_models.py
- audio.py
- expert.py
- hubconf.py
- __init__.py
- builder.py
- expert.py
- hubconf.py
- __init__.py
- expert.py
- hubconf.py
- __init__.py
- audio.py
- expert.py
- hubconf.py
- vggish.py
- vggish_params.py
- __init__.py
- expert.py
- hubconf.py
- __init__.py
- convert.py
- expert.py
- hubconf.py
- __init__.py
- convert.py
- expert.py
- hubconf.py
- wav2vec_model.py
- __init__.py
- convert.py
- expert.py
- hubconf.py
- wav2vec2_model.py
- __init__.py
- expert.py
- hubconf.py
- modules.py
- WavLM.py
- __init__.py
- interfaces.py
- README.md
- utils.py
- __init__.py
- audio_info.py
- benchmark.py
- download.py
- override.py
- pseudo_data.py
- seed.py
- __init__.py
- add_config_to_ckpt.py
- allclose.py
- audio.py
- check_hub.py
- compare_wav2vec2.py
- compute_acc.py
- data.py
- download.py
- extract_pase.py
- extract_sample.py
- extract_settings.py
- fix_ckpt.py
- get_best_dev.py
- get_best_score.py
- helper.py
- observe_ckpt.py
- observe_input.py
- observe_lnsr.py
- observe_speaker.py
- observe_weights.py
- overwrite_settings.py
- print_settings.py
- print_url_cache_path.py
- run_sig_test.py
- timer.py
- visualize_weight.py
- __init__.py
- hub.py
- main.py
- optimizers.py
- run_downstream.py
- run_pretrain.py
- run_while.sh
- schedulers.py
- version.txt
- fbank.conf
- cmd.sh
- compute_fmllr.sh
- decode.sh
- dump_fbank_cmvn.sh
- dump_fmllr_cmvn.sh
- dump_mfcc_cmvn.sh
- run.sh
- cmd.sh
- dump_fmllr_cmvn.sh
- run.sh
- libri_transformer_fmllr_ft.cfg
- libri_transformer_liGRU_fmllr.cfg
- libri_transformer_liGRU_fmllr_ft.cfg
- libri_transformer_liGRU_mfcc.cfg
- timit_transformer_fmllr_ft.cfg
- timit_transformer_liGRU_fmllr.cfg
- timit_transformer_liGRU_fmllr_best.cfg
- timit_transformer_liGRU_fmllr_deep.cfg
- Lin.proto
- Transformer.proto
- find_lowest_wer.py
- nn_transformer.py
- example_extract_finetune.py
- example_solver.py
- runner.py
- tutorial_use_pretrained_model_without_preprocessing.py
- test_superb.py
- conftest.py
- test_audio_info.py
- test_balanced_weighted_sampler.py
- test_beam_decoder.py
- test_download.py
- test_er.py
- test_fluent_commands.py
- test_frame_label.py
- test_g2p.py
- test_librispeech.py
- test_linear.py
- test_load_audio.py
- test_metric.py
- test_pooling.py
- test_quesst14.py
- test_rnn.py
- test_sampler.py
- test_snips.py
- test_sorted_sampler.py
- test_sox.py
- test_specaug_model.py
- test_speech_commands.py
- test_tokenizer.py
- test_upstream.py
- test_version.py
- test_vocabulary.py
- test_voxceleb1sid.py
- extract_feat.py
- assert_python_version.py
- download_url_and_show_path.py
- extract_feat.py
- load_ckpt.py
- print_pkl.py
- test_upstream.py
- .dockerignore
- .env
- .gitignore
- .pre-commit-config.yaml
- Dockerfile
- find_content.sh
- hubconf.py
- LICENSE
- pyrightconfig.json
- pytest.ini
- README.md
- setup.cfg
- setup.py
- tox.ini
- valid_paths.txt
# Installation Guide
1. Get the code
git clone https://github.com/s3prl/s3prl
Downloads the entire project code from GitHub to your computer.
cd s3prl
Moves into the project folder you just downloaded.
2. Official Install Script
Easy RecommendedPrerequisites
- Python 3 Python is required to use pip.
pip install s3prl
Installs the package published on PyPI directly β no need to clone the source.
After installing, open a new terminal and run the program's version command (e.g. --version) to confirm it worked.
Pulled directly from this repo's README.
3. Docker
EasyPrerequisites
- Git Needed to download the project code from GitHub.
- Docker Desktop Needed to build and run containers. Install it and keep it running in the background.
docker build -t s3prl .
Builds a runnable image based on the Dockerfile.
docker run -p 8080:80 s3prl
Runs the built image as an actual container.
Run docker compose ps to check the containers are Up. If the README mentions a port, open http://localhost:PORT in your browser.
4. Python
EasyPrerequisites
pip install s3prl
Installs the package published on PyPI directly β no need to clone the source.
If it runs without errors and prints output in the terminal, it worked.
Pulled directly from this repo's README.
// repository documentation
Was this content helpful?
(0 ratings)
