FlashSpeech
ACM MM 2024 FlashSpeech: Efficient Zero-Shot Speech Synthesis
File Explorer
Download Latest Version (.zip)- inference.py
- preprocess.py
- train_new.py
- calc_metrics.py
- base.json
- ns2.json
- transformer.json
- tts.json
- docker.md
- README.md
- exp_config.json
- exp_config_base.json
- exp_config_base_s1.json
- exp_config_base_s2.json
- exp_config_s1.json
- exp_config_s2.json
- run_inference.sh
- run_slurm.sh
- run_slurm2.sh
- run_train.sh
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- base_dataset.cpython-310.pyc
- base_dataset.cpython-39.pyc
- base_sampler.cpython-310.pyc
- base_sampler.cpython-39.pyc
- base_trainer.cpython-310.pyc
- base_trainer.cpython-39.pyc
- new_inference.cpython-310.pyc
- new_inference.cpython-39.pyc
- new_trainer.cpython-310.pyc
- new_trainer.cpython-39.pyc
- __init__.py
- base_dataset.py
- base_inference.py
- base_sampler.py
- base_trainer.py
- new_dataset.py
- new_inference.py
- new_trainer.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- tts_dataset.cpython-310.pyc
- tts_dataset.cpython-39.pyc
- tts_inferece.cpython-310.pyc
- tts_trainer.cpython-310.pyc
- tts_trainer.cpython-39.pyc
- __init__.py
- tts_dataset.py
- tts_inferece.py
- tts_trainer.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- fs2.cpython-310.pyc
- fs2.cpython-39.pyc
- fs2_dataset.cpython-310.pyc
- fs2_dataset.cpython-39.pyc
- fs2_inference.cpython-310.pyc
- fs2_trainer.cpython-310.pyc
- fs2_trainer.cpython-39.pyc
- __init__.py
- fs2.py
- fs2_dataset.py
- fs2_inference.py
- fs2_trainer.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- alignments.cpython-310.pyc
- alignments.cpython-39.pyc
- jets.cpython-310.pyc
- jets.cpython-39.pyc
- jets_dataset.cpython-310.pyc
- jets_dataset.cpython-39.pyc
- jets_inference.cpython-310.pyc
- jets_loss.cpython-310.pyc
- jets_loss.cpython-39.pyc
- jets_trainer.cpython-310.pyc
- jets_trainer.cpython-39.pyc
- length_regulator.cpython-310.pyc
- length_regulator.cpython-39.pyc
- __init__.py
- alignments.py
- jets.py
- jets_dataset.py
- jets_inference.py
- jets_loss.py
- jets_trainer.py
- length_regulator.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- diffusion.cpython-310.pyc
- diffusion.cpython-39.pyc
- diffusion_flow.cpython-310.pyc
- diffusion_flow.cpython-39.pyc
- flashspeech.cpython-310.pyc
- flashspeech_inference.cpython-310.pyc
- flashspeech_trainer.cpython-310.pyc
- flashspeech_trainer_stage2.cpython-310.pyc
- ict.cpython-310.pyc
- ns2.cpython-310.pyc
- ns2.cpython-39.pyc
- ns2_dataset.cpython-310.pyc
- ns2_dataset.cpython-39.pyc
- ns2_inference.cpython-310.pyc
- ns2_loss.cpython-310.pyc
- ns2_loss.cpython-39.pyc
- ns2_trainer.cpython-310.pyc
- ns2_trainer.cpython-39.pyc
- prior_encoder.cpython-310.pyc
- prior_encoder.cpython-39.pyc
- wavenet.cpython-310.pyc
- wavenet.cpython-39.pyc
- wavlm_loss.cpython-310.pyc
- x_codec_baseline.cpython-39.pyc
- x_codec_baseline.cpython-310.pyc
- x_codec_baseline.cpython-39.pyc
- x_codec_baseline.py
- __init__.py
- diffusion.py
- diffusion_flow.py
- flashspeech.py
- flashspeech_inference.py
- flashspeech_inference_librispeech_test_clean.py
- flashspeech_trainer.py
- flashspeech_trainer_old.py
- flashspeech_trainer_stage2.py
- ict.py
- ns2.py
- ns2_dataset.py
- ns2_dataset_old.py
- ns2_inference.py
- ns2_loss.py
- ns2_trainer.py
- prior_encoder.py
- wavenet.py
- wavlm_loss.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- valle.cpython-310.pyc
- valle.cpython-39.pyc
- valle_dataset.cpython-310.pyc
- valle_dataset.cpython-39.pyc
- valle_inference.cpython-310.pyc
- valle_trainer.cpython-310.pyc
- valle_trainer.cpython-39.pyc
- __init__.py
- valle.py
- valle_dataset.py
- valle_inference.py
- valle_trainer.py
- base_trainer.cpython-310.pyc
- base_trainer.cpython-39.pyc
- valle_ar_trainer.cpython-310.pyc
- valle_ar_trainer.cpython-39.pyc
- valle_nar_trainer.cpython-310.pyc
- valle_nar_trainer.cpython-39.pyc
- base_trainer.py
- g2p_processor.py
- libritts_dataset.py
- modeling_llama.py
- valle_ar.py
- valle_ar_trainer.py
- valle_collator.py
- valle_inference.py
- valle_nar.py
- valle_nar_trainer.py
- __init__.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- gated_activation_unit.cpython-310.pyc
- gated_activation_unit.cpython-39.pyc
- snake.cpython-310.pyc
- snake.cpython-39.pyc
- __init__.py
- gated_activation_unit.py
- snake.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- act.cpython-310.pyc
- act.cpython-39.pyc
- filter.cpython-310.pyc
- filter.cpython-39.pyc
- resample.cpython-310.pyc
- resample.cpython-39.pyc
- __init__.py
- act.py
- filter.py
- resample.py
- base_module.cpython-310.pyc
- base_module.cpython-39.pyc
- base_module.py
- __init__.py
- base.py
- dac.py
- discriminator.py
- encodec.py
- __init__.py
- layers.py
- loss.py
- quantize.py
- __init__.py
- bidilated_conv.py
- residual_block.py
- karras_diffusion.py
- random_utils.py
- sample.py
- attention.py
- basic.py
- resblock.py
- unet.py
- __init__.py
- __init__.py
- distributions.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- standard_duration_predictor.cpython-310.pyc
- standard_duration_predictor.cpython-39.pyc
- stochastic_duration_predictor.cpython-310.pyc
- stochastic_duration_predictor.cpython-39.pyc
- __init__.py
- standard_duration_predictor.py
- stochastic_duration_predictor.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- token_encoder.cpython-310.pyc
- token_encoder.cpython-39.pyc
- __init__.py
- condition_encoder.py
- position_encoder.py
- token_encoder.py
- modules.cpython-310.pyc
- modules.cpython-39.pyc
- modules.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- input_strategies.cpython-310.pyc
- input_strategies.cpython-39.pyc
- scaling.cpython-310.pyc
- scaling.cpython-39.pyc
- utils.cpython-310.pyc
- utils.cpython-39.pyc
- __init__.py
- input_strategies.py
- scaling.py
- utils.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- core.cpython-39-x86_64-linux-gnu.so
- core.o
- core.cpython-39-x86_64-linux-gnu.so
- __init__.py
- core.c
- core.pyx
- setup.py
- transformers.cpython-310.pyc
- transformers.cpython-39.pyc
- transformers.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- sine_excitation.cpython-310.pyc
- sine_excitation.cpython-39.pyc
- __init__.py
- sine_excitation.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- norm.cpython-310.pyc
- norm.cpython-39.pyc
- __init__.py
- norm.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- attentions.cpython-310.pyc
- attentions.cpython-39.pyc
- Layers.cpython-310.pyc
- Layers.cpython-39.pyc
- mh_attention.cpython-310.pyc
- mh_attention.cpython-39.pyc
- Models.cpython-310.pyc
- Models.cpython-39.pyc
- Modules.cpython-310.pyc
- Modules.cpython-39.pyc
- position_embedding.cpython-310.pyc
- position_embedding.cpython-39.pyc
- SubLayers.cpython-310.pyc
- SubLayers.cpython-39.pyc
- transformer.cpython-310.pyc
- transformer.cpython-39.pyc
- transforms.cpython-310.pyc
- transforms.cpython-39.pyc
- __init__.py
- attentions.py
- Constants.py
- Layers.py
- mh_attention.py
- Models.py
- Modules.py
- position_embedding.py
- SubLayers.py
- transformer.py
- transforms.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- gan_utils.cpython-310.pyc
- gan_utils.cpython-39.pyc
- norm2d.cpython-310.pyc
- norm2d.cpython-39.pyc
- __init__.py
- gan_utils.py
- norm2d.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- predictor.cpython-310.pyc
- predictor.cpython-39.pyc
- predictor.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- attention.cpython-310.pyc
- attention.cpython-39.pyc
- convolution.cpython-310.pyc
- convolution.cpython-39.pyc
- encoder.cpython-310.pyc
- encoder.cpython-39.pyc
- encoder_layer.cpython-310.pyc
- encoder_layer.cpython-39.pyc
- subsampling.cpython-310.pyc
- subsampling.cpython-39.pyc
- __init__.py
- attention.py
- convolution.py
- encoder.py
- encoder_layer.py
- subsampling.py
- paraformer.cpython-310.pyc
- paraformer.cpython-39.pyc
- utils.cpython-310.pyc
- utils.cpython-39.pyc
- beam_search.cpython-310.pyc
- beam_search.cpython-39.pyc
- ctc.cpython-310.pyc
- ctc.cpython-39.pyc
- ctc_prefix_score.cpython-310.pyc
- ctc_prefix_score.cpython-39.pyc
- scorer_interface.cpython-310.pyc
- scorer_interface.cpython-39.pyc
- beam_search.py
- ctc.py
- ctc_prefix_score.py
- scorer_interface.py
- paraformer.py
- utils.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- attention.cpython-310.pyc
- attention.cpython-39.pyc
- conv2d.cpython-310.pyc
- conv2d.cpython-39.pyc
- convolution.cpython-310.pyc
- convolution.cpython-39.pyc
- encoder.cpython-310.pyc
- encoder.cpython-39.pyc
- encoder_layer.cpython-310.pyc
- encoder_layer.cpython-39.pyc
- positionwise_feed_forward.cpython-310.pyc
- positionwise_feed_forward.cpython-39.pyc
- subsampling.cpython-310.pyc
- subsampling.cpython-39.pyc
- __init__.py
- attention.py
- conv2d.py
- convolution.py
- encoder.py
- encoder_layer.py
- positionwise_feed_forward.py
- subsampling.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- joint.cpython-310.pyc
- joint.cpython-39.pyc
- predictor.cpython-310.pyc
- predictor.cpython-39.pyc
- transducer.cpython-310.pyc
- transducer.cpython-39.pyc
- greedy_search.cpython-310.pyc
- greedy_search.cpython-39.pyc
- prefix_beam_search.cpython-310.pyc
- prefix_beam_search.cpython-39.pyc
- greedy_search.py
- prefix_beam_search.py
- __init__.py
- joint.py
- predictor.py
- transducer.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- asr_model.cpython-310.pyc
- asr_model.cpython-39.pyc
- attention.cpython-310.pyc
- attention.cpython-39.pyc
- cmvn.cpython-310.pyc
- cmvn.cpython-39.pyc
- convolution.cpython-310.pyc
- convolution.cpython-39.pyc
- ctc.cpython-310.pyc
- ctc.cpython-39.pyc
- decoder.cpython-310.pyc
- decoder.cpython-39.pyc
- decoder_layer.cpython-310.pyc
- decoder_layer.cpython-39.pyc
- embedding.cpython-310.pyc
- embedding.cpython-39.pyc
- encoder.cpython-310.pyc
- encoder.cpython-39.pyc
- encoder_layer.cpython-310.pyc
- encoder_layer.cpython-39.pyc
- label_smoothing_loss.cpython-310.pyc
- label_smoothing_loss.cpython-39.pyc
- positionwise_feed_forward.cpython-310.pyc
- positionwise_feed_forward.cpython-39.pyc
- subsampling.cpython-310.pyc
- subsampling.cpython-39.pyc
- __init__.py
- asr_model.py
- attention.py
- cmvn.py
- convolution.py
- ctc.py
- decoder.py
- decoder_layer.py
- embedding.py
- encoder.py
- encoder_layer.py
- label_smoothing_loss.py
- positionwise_feed_forward.py
- subsampling.py
- swish.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- checkpoint.cpython-310.pyc
- checkpoint.cpython-39.pyc
- cmvn.cpython-310.pyc
- cmvn.cpython-39.pyc
- common.cpython-310.pyc
- common.cpython-39.pyc
- init_model.cpython-310.pyc
- init_model.cpython-39.pyc
- mask.cpython-310.pyc
- mask.cpython-39.pyc
- __init__.py
- checkpoint.py
- cmvn.py
- common.py
- config.py
- ctc_util.py
- executor.py
- file_utils.py
- init_model.py
- mask.py
- scheduler.py
- __init__.py
- README.md
- __init__.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- optimizers.cpython-310.pyc
- optimizers.cpython-39.pyc
- __init__.py
- optimizers.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- cdmusiceval.cpython-39.pyc
- coco.cpython-39.pyc
- cocoeval.cpython-39.pyc
- csd.cpython-39.pyc
- customsvcdataset.cpython-39.pyc
- hifitts.cpython-39.pyc
- kising.cpython-39.pyc
- librilight.cpython-39.pyc
- libritts.cpython-39.pyc
- lijian.cpython-39.pyc
- ljspeech.cpython-39.pyc
- ljspeech_vocoder.cpython-39.pyc
- m4singer.cpython-39.pyc
- metadata.cpython-310.pyc
- metadata.cpython-39.pyc
- nus48e.cpython-39.pyc
- opencpop.cpython-39.pyc
- opensinger.cpython-39.pyc
- opera.cpython-39.pyc
- pjs.cpython-39.pyc
- popbutfy.cpython-39.pyc
- popcs.cpython-39.pyc
- processor.cpython-39.pyc
- svcc.cpython-39.pyc
- svcceval.cpython-39.pyc
- vctk.cpython-39.pyc
- vctksample.cpython-39.pyc
- vocalist.cpython-39.pyc
- __init__.py
- dnsmos.py
- separate_fast.py
- silero_vad.py
- whisper_asr.py
- __init__.py
- logger.py
- tool.py
- config.json
- env.sh
- main.py
- README.md
- requirements.txt
- __init__.py
- bigdata.py
- cdmusiceval.py
- coco.py
- cocoeval.py
- csd.py
- customsvcdataset.py
- hifitts.py
- kising.py
- librilight.py
- libritts.py
- lijian.py
- ljspeech.py
- ljspeech_vocoder.py
- m4singer.py
- metadata.py
- nus48e.py
- opencpop.py
- opensinger.py
- opera.py
- pjs.py
- popbutfy.py
- popcs.py
- processor.py
- svcc.py
- svcceval.py
- vctk.py
- vctkfewsinger.py
- vctksample.py
- vocalist.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- acoustic_extractor.cpython-310.pyc
- acoustic_extractor.cpython-39.pyc
- content_extractor.cpython-310.pyc
- content_extractor.cpython-39.pyc
- data_augment.cpython-39.pyc
- phone_extractor.cpython-310.pyc
- phone_extractor.cpython-39.pyc
- __init__.py
- acoustic_extractor.py
- audio_features_extractor.py
- content_extractor.py
- data_augment.py
- descriptive_text_features_extractor.py
- phone_extractor.py
- text_features_extractor.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- scheduler.cpython-310.pyc
- scheduler.cpython-39.pyc
- __init__.py
- scheduler.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- cleaners.cpython-310.pyc
- cleaners.cpython-39.pyc
- cmudict.cpython-310.pyc
- cmudict.cpython-39.pyc
- g2p.cpython-310.pyc
- g2p_module.cpython-310.pyc
- g2p_module.cpython-39.pyc
- numbers.cpython-310.pyc
- numbers.cpython-39.pyc
- pinyin.cpython-310.pyc
- pinyin.cpython-39.pyc
- symbol_table.cpython-310.pyc
- symbol_table.cpython-39.pyc
- symbols.cpython-310.pyc
- symbols.cpython-39.pyc
- text_token_collation.cpython-310.pyc
- text_token_collation.cpython-39.pyc
- librispeech-lexicon.txt
- pinyin-lexicon-r.txt
- __init__.py
- cleaners.py
- cmudict.py
- g2p.py
- g2p_module.py
- numbers.py
- pinyin.py
- symbol_table.py
- symbols.py
- text_token_collation.py
- __init__.cpython-310.pyc
- __init__.cpython-39.pyc
- audio.cpython-310.pyc
- audio.cpython-39.pyc
- audio_slicer.cpython-310.pyc
- audio_slicer.cpython-39.pyc
- cut_by_vad.cpython-39.pyc
- data_utils.cpython-310.pyc
- data_utils.cpython-39.pyc
- dsp.cpython-310.pyc
- dsp.cpython-39.pyc
- duration.cpython-39.pyc
- f0.cpython-39.pyc
- hparam.cpython-310.pyc
- hparam.cpython-39.pyc
- io.cpython-310.pyc
- io.cpython-39.pyc
- io_optim.cpython-310.pyc
- io_optim.cpython-39.pyc
- mel.cpython-310.pyc
- mel.cpython-39.pyc
- mfa_prepare.cpython-39.pyc
- stft.cpython-310.pyc
- stft.cpython-39.pyc
- tokenizer.cpython-310.pyc
- tokenizer.cpython-39.pyc
- topk_sampling.cpython-310.pyc
- topk_sampling.cpython-39.pyc
- util.cpython-310.pyc
- util.cpython-39.pyc
- whisper_transcription.cpython-39.pyc
- world.cpython-39.pyc
- __init__.py
- hps.py
- __init__.py
- audio.py
- audio_slicer.py
- cut_by_vad.py
- data_utils.py
- distribution.py
- dsp.py
- duration.py
- f0.py
- hparam.py
- hubert.py
- io.py
- io_optim.py
- mel.py
- mert.py
- mfa_prepare.py
- model_summary.py
- prompt_preparer.py
- ssim.py
- stft.py
- symbol_table.py
- tokenizer.py
- topk_sampling.py
- trainer_utils.py
- util.py
- whisper_transcription.py
- world.py
- env.sh
- README.md
# Installation Guide
1. Get the code
git clone https://github.com/zhenye234/FlashSpeech
Downloads the entire project code from GitHub to your computer.
cd FlashSpeech
Moves into the project folder you just downloaded.
2. Python
Easy RecommendedPrerequisites
β οΈ This is a large repository, so this method may point to an internal sub-package rather than the actual core product. Check the full README as well.
pip install -r preprocessors/Emilia/requirements.txt
Installs the Python libraries listed in requirements.txt (or similar).
python <μ€νν νμΌλͺ
>.py # READMEμμ μ νν μ€ν νμΌλͺ
μ νμΈνμΈμ
Runs the Python script (or module).
If it runs without errors and prints output in the terminal, it worked.
// repository documentation
Was this content helpful?
(0 ratings)
