learnable-speech
This repo is text to speech with learnable audio encoder without alignment with transcript reference
File Explorer
Download Latest Version (.zip)- image.png
- __init__.cpython-310.pyc
- __init__.cpython-310.pyc
- audio_signal.cpython-310.pyc
- display.cpython-310.pyc
- dsp.cpython-310.pyc
- effects.cpython-310.pyc
- ffmpeg.cpython-310.pyc
- loudness.cpython-310.pyc
- playback.cpython-310.pyc
- util.cpython-310.pyc
- whisper.cpython-310.pyc
- __init__.cpython-310.pyc
- __init__.py
- headers.html
- pandoc.css
- widget.html
- __init__.py
- audio_signal.py
- display.py
- dsp.py
- effects.py
- ffmpeg.py
- loudness.py
- playback.py
- util.py
- whisper.py
- __init__.cpython-310.pyc
- datasets.cpython-310.pyc
- preprocess.cpython-310.pyc
- transforms.cpython-310.pyc
- __init__.py
- datasets.py
- preprocess.py
- transforms.py
- __init__.cpython-310.pyc
- distance.cpython-310.pyc
- quality.cpython-310.pyc
- spectral.cpython-310.pyc
- __init__.py
- distance.py
- quality.py
- spectral.py
- __init__.cpython-310.pyc
- accelerator.cpython-310.pyc
- decorators.cpython-310.pyc
- experiment.cpython-310.pyc
- __init__.cpython-310.pyc
- base.cpython-310.pyc
- spectral_gate.cpython-310.pyc
- __init__.py
- base.py
- spectral_gate.py
- __init__.py
- accelerator.py
- decorators.py
- experiment.py
- __init__.py
- post.py
- preference.py
- log.txt
- base.yml
- config.yml
- configx2.yml
- base.py
- extract.sh
- extract_dac_latents.py
- inference.py
- layers.py
- loss.py
- model.py
- README.md
- requirements.txt
- run.sh
- train.py
- dae.yaml
- imagenet_ae.yaml
- imagenet_zdm.yaml
- dito-B-audio.yaml
- dito-B-f8c4-noise-sync.yaml
- dito-B-f8c4.yaml
- dito-L-f8c4.yaml
- dito-XL-f8c4-noise-sync.yaml
- dito-XL-f8c4.yaml
- eval50k_zdm-XL_dito-XL-f8c4-noise-sync.yaml
- eval50k_zdm-XL_dito-XL-f8c4.yaml
- zdm-XL_dito-XL-f8c4-noise-sync.yaml
- zdm-XL_dito-XL-f8c4.yaml
- zdm-XL_imagenet.yaml
- dito.yaml
- glpto.yaml
- zdm.yaml
- __init__.py
- class_folder.py
- class_folder_audio.py
- datasets.py
- image_folder.py
- webdataset.py
- wrapper_audio_cae.py
- wrapper_cae.py
- dito.png
- wandb.yaml
- __init__.py
- fm.py
- samplers.py
- __init__.py
- headers.html
- pandoc.css
- widget.html
- __init__.py
- audio_signal.py
- display.py
- dsp.py
- effects.py
- ffmpeg.py
- loudness.py
- playback.py
- util.py
- whisper.py
- __init__.py
- datasets.py
- preprocess.py
- transforms.py
- __init__.py
- distance.py
- quality.py
- spectral.py
- __init__.py
- base.py
- spectral_gate.py
- __init__.py
- accelerator.py
- decorators.py
- experiment.py
- __init__.py
- post.py
- preference.py
- __init__.py
- base.py
- layers.py
- loss.py
- model.py
- utils.py
- __init__.py
- discriminator.py
- lpips.py
- model.py
- quantizer.py
- utils.py
- __init__.py
- dito.py
- glpto.py
- ldm_base.py
- renderers.py
- __init__.py
- consistency_audio_decoder_unet.py
- consistency_decoder_unet.py
- dit.py
- __init__.py
- models.py
- __init__.py
- audio_ldm_trainer.py
- base_trainer.py
- ldm_trainer.py
- trainers.py
- __init__.py
- geometry.py
- utils.py
- .gitignore
- audio_dito_inference.py
- image_dito_inference.py
- requirements.txt
- run.py
- run.sh
- download_pretrained.sh
- prepare_data.sh
- train_full_pipeline.sh
- training_configs.sh
- upload_to_hf.py
- dingding.png
- export_jit.py
- export_onnx.py
- __init__.py
- cosyvoice.py
- frontend.py
- model.py
- __init__.py
- dataset.py
- processor.py
- decoder.py
- flow.py
- flow_matching.py
- length_regulator.py
- discriminator.py
- f0_predictor.py
- generator.py
- hifigan.py
- llm.py
- multilingual_zh_ja_yue_char_del.tiktoken
- tokenizer.py
- __init__.py
- activation.py
- arch_util.py
- attention.py
- convolution.py
- decoder.py
- decoder_layer.py
- embedding.py
- encoder.py
- encoder_layer.py
- label_smoothing_loss.py
- positionwise_feed_forward.py
- subsampling.py
- upsample_encoder.py
- xtransformers.py
- __init__.py
- class_utils.py
- common.py
- executor.py
- file_utils.py
- frontend_utils.py
- losses.py
- mask.py
- scheduler.py
- train_utils.py
- __init__.py
- download_and_untar.sh
- prepare_data.py
- __init__.py
- config.py
- denoiser.py
- env.py
- LICENSE
- meldataset.py
- models.py
- README.md
- xutils.py
- __init__.py
- decoder.py
- flow_matching.py
- text_encoder.py
- transformer.py
- __init__.py
- baselightningmodule.py
- matcha_tts.py
- __init__.py
- export.py
- infer.py
- __init__.py
- cleaners.py
- numbers.py
- symbols.py
- __init__.py
- core.pyx
- setup.py
- __init__.py
- audio.py
- generate_data_statistics.py
- instantiators.py
- logging_utils.py
- model.py
- pylogger.py
- rich_utils.py
- utils.py
- __init__.py
- app.py
- cli.py
- python-publish.yml
- unit_test_cpu.yaml
- mel_filters.npz
- __init__.py
- cli.py
- model.py
- model_v2.py
- utils.py
- test_batch_efficiency.py
- test_onnx.py
- .flake8
- .gitignore
- .pre-commit-config.yaml
- LICENSE
- MANIFEST.in
- README.md
- requirements.txt
- setup.py
- create_data_list.py
- download_dataset.py
- download_vivoice.py
- extract_embedding.py
- extract_speech_token.py
- generate_json_index.py
- inv_file_processor.py
- make_parquet_list.py
- validate_data.py
- .gitmodules
- config.yaml
- config_hf.yaml
- data_test.list
- dev.ipynb
- files_test.txt
- flow_run.sh
- inference.py
- llm_run.sh
- test_train.sh
- train.py
- .dockerignore
- app.py
- Dockerfile
- README.md
- requirements.txt
- TRAINING_GUIDE.md
# Installation Guide
git clone https://github.com/primepake/learnable-speech
Downloads the entire project code from GitHub to your computer.
cd learnable-speech
Moves into the project folder you just downloaded.
2. Official Install Script
Easy Recommended- Python 3 Python is required to use pip.
pip install s3tokenizer
Installs the package published on PyPI directly β no need to clone the source.
pip3 install .
Installs the package published on PyPI directly β no need to clone the source.
Pulled directly from this repo's README.
3. Docker
Easy- Git Needed to download the project code from GitHub.
- Docker Desktop Needed to build and run containers. Install it and keep it running in the background.
docker build -t learnable-speech .
Builds a runnable image based on the Dockerfile.
docker run -p 8080:80 learnable-speech
Runs the built image as an actual container.
4. Python
Easypip install -r requirements.txt
Installs the Python libraries listed in requirements.txt (or similar).
pip install s3tokenizer
Installs the package published on PyPI directly β no need to clone the source.
pip3 install .
Installs the package published on PyPI directly β no need to clone the source.
Pulled directly from this repo's README.
