mlx-AuK
AuK: Native Apple Silicon MLX Port for 1.5B Speech Foundation Model (TTS, Vocal/Content Edit, Enhancement & Separation)
File Explorer
- accent-dongbei-input.wav
- accent-fujian-input.wav
- accent-hubei-input.wav
- accent-hunan-input.wav
- accent-india-input.wav
- accent-japan-input.wav
- accent-sichuan-input.wav
- accent-tibetan-input.wav
- ce-en-1-input.wav
- ce-en-2-input.wav
- ce-zh-1-input.wav
- ce-zh-2-input.wav
- emotion-edit_en-1-input.wav
- emotion-edit_en-2-input.wav
- emotion-edit_zh-1-input.wav
- emotion-edit_zh-2-input.wav
- en-1-input.wav
- en-2-input.wav
- en-c-input.wav
- en-d-input.wav
- energy-edit-1-input.wav
- energy-edit-2-input.wav
- pitch-1-input.wav
- pitch-2-input.wav
- se-en-1-input.wav
- se-en-2-input.wav
- se-zh-1-input.wav
- se-zh-2-input.wav
- speed-edit-1-input.wav
- speed-edit-input.wav
- sr-en-1-input.wav
- sr-en-2-input.wav
- sr-zh-1-input.wav
- sr-zh-2-input.wav
- vc-1-input.wav
- vc-2-input.wav
- vc-3-input.wav
- vc-4-input.wav
- vocal-1-input.wav
- vocal-2-input.wav
- vocal-3-input.wav
- vocal-4-input.wav
- vocaledit-en-1-input.wav
- vocaledit-en-2-input.wav
- vocaledit-zh-1-input.wav
- vocaledit-zh-2-input.wav
- wh-n2w-en-input.wav
- wh-n2w-zh-input.wav
- wh-w2n-en-input.wav
- wh-w2n-zh-input.wav
- zh-1-input.wav
- zh-2-input.wav
- zh-a-input.wav
- zh-b-input.wav
- zs-tts-1-input.wav
- zs-tts-2-input.wav
- zs-tts-3-input.wav
- zs-tts-4-input.wav
- accent-dongbei-output.wav
- accent-fujian-output.wav
- accent-hubei-output.wav
- accent-hunan-output.wav
- accent-india-output.wav
- accent-japan-output.wav
- accent-sichuan-output.wav
- accent-tibetan-output.wav
- ce-en-1-add.wav
- ce-en-1-change.wav
- ce-en-1-delete.wav
- ce-en-1-input.wav
- ce-en-2-add.wav
- ce-en-2-change.wav
- ce-en-2-delete.wav
- ce-en-2-input.wav
- ce-zh-1-add.wav
- ce-zh-1-change.wav
- ce-zh-1-delete.wav
- ce-zh-1-input.wav
- ce-zh-2-add.wav
- ce-zh-2-change.wav
- ce-zh-2-delete.wav
- ce-zh-2-input.wav
- emotion-edit_en-1-input.wav
- emotion-edit_en-2-input.wav
- emotion-edit_zh-1-input.wav
- emotion-edit_zh-2-input.wav
- en-1-afraid.wav
- en-1-angry.wav
- en-1-content.wav
- en-1-happy.wav
- en-1-input.wav
- en-1-number.wav
- en-1-sad.wav
- en-2-afraid.wav
- en-2-angry.wav
- en-2-content.wav
- en-2-happy.wav
- en-2-input.wav
- en-2-number.wav
- en-2-sad.wav
- en-c-breath.wav
- en-c-input.wav
- en-c-pause.wav
- en-c-sneeze.wav
- en-d-cry.wav
- en-d-hiss.wav
- en-d-hmm.wav
- en-d-input.wav
- energy-edit-1-+10.wav
- energy-edit-1-+15.wav
- energy-edit-1-+5.wav
- energy-edit-1--10.wav
- energy-edit-1--15.wav
- energy-edit-1--5.wav
- energy-edit-1-input.wav
- energy-edit-2-+10.wav
- energy-edit-2-+15.wav
- energy-edit-2-+5.wav
- energy-edit-2--10.wav
- energy-edit-2--15.wav
- energy-edit-2--5.wav
- energy-edit-2-input.wav
- instruct-tts-1-output.wav
- instruct-tts-2-output.wav
- instruct-tts-3-output.wav
- instruct-tts-4-output.wav
- pitch-1-+1.wav
- pitch-1-+2.wav
- pitch-1-+3.wav
- pitch-1--1.wav
- pitch-1--2.wav
- pitch-1--3.wav
- pitch-1-input.wav
- pitch-2-+1.wav
- pitch-2-+2.wav
- pitch-2-+3.wav
- pitch-2--1.wav
- pitch-2--2.wav
- pitch-2--3.wav
- pitch-2-input.wav
- se-en-1-output.wav
- se-en-2-output.wav
- se-zh-1-output.wav
- se-zh-2-output.wav
- speed-edit-0.5.wav
- speed-edit-0.75.wav
- speed-edit-1-0.5.wav
- speed-edit-1-0.75.wav
- speed-edit-1-1.25.wav
- speed-edit-1-1.5.wav
- speed-edit-1-2.wav
- speed-edit-1-input.wav
- speed-edit-1.25.wav
- speed-edit-1.5.wav
- speed-edit-2.wav
- speed-edit-input.wav
- sr-en-1-output.wav
- sr-en-2-output.wav
- sr-zh-1-output.wav
- sr-zh-2-output.wav
- vc-1-output.wav
- vc-2-output.wav
- vc-3-output.wav
- vc-4-output.wav
- vocal-1-output.wav
- vocal-2-output.wav
- vocal-3-output.wav
- vocal-4-output.wav
- vocaledit-en-1-output.wav
- vocaledit-en-2-output.wav
- vocaledit-zh-1-output.wav
- vocaledit-zh-2-output.wav
- wh-n2w-en-output.wav
- wh-n2w-zh-output.wav
- wh-w2n-en-output.wav
- wh-w2n-zh-output.wav
- zh-1-afraid.wav
- zh-1-calm.wav
- zh-1-content.wav
- zh-1-happy.wav
- zh-1-input.wav
- zh-1-number.wav
- zh-1-sad.wav
- zh-2-afraid.wav
- zh-2-angry.wav
- zh-2-content.wav
- zh-2-happy.wav
- zh-2-input.wav
- zh-2-number.wav
- zh-2-sad.wav
- zh-a-cough.wav
- zh-a-filler.wav
- zh-a-input.wav
- zh-a-sigh.wav
- zh-b-inhale.wav
- zh-b-input.wav
- zh-b-laugh.wav
- zh-b-surprise.wav
- zs-tts-1-output.wav
- zs-tts-2-output.wav
- zs-tts-3-output.wav
- zs-tts-4-output.wav
- .DS_Store
- accent-edit_accent-dongbei.wav
- accent-edit_accent-fujian.wav
- accent-edit_accent-hubei.wav
- accent-edit_accent-hunan.wav
- accent-edit_accent-india.wav
- accent-edit_accent-japan.wav
- accent-edit_accent-sichuan.wav
- accent-edit_accent-tibetan.wav
- ce-en-1_add.wav
- ce-en-1_delete.wav
- ce-en-1_replace.wav
- ce-en-2_add.wav
- ce-en-2_delete.wav
- ce-en-2_replace.wav
- ce-zh-1_add.wav
- ce-zh-1_delete.wav
- ce-zh-1_replace.wav
- ce-zh-2_add.wav
- ce-zh-2_delete.wav
- ce-zh-2_replace.wav
- content-edit-speech_ce-en-1.wav
- content-edit-speech_ce-en-2.wav
- content-edit-speech_ce-zh-1.wav
- content-edit-speech_ce-zh-2.wav
- emotion-edit_emo-en-1.wav
- emotion-edit_emo-en-2.wav
- emotion-edit_emo-zh-1.wav
- emotion-edit_emo-zh-2.wav
- en-1_content.wav
- en-1_number.wav
- en-2_content.wav
- en-2_number.wav
- enhance-speech_se-en-1.wav
- enhance-speech_se-en-2.wav
- enhance-speech_se-zh-1.wav
- enhance-speech_se-zh-2.wav
- extract-vocals_ev-1.wav
- extract-vocals_ev-2.wav
- extract-vocals_ev-3.wav
- extract-vocals_ev-4.wav
- instruct-tts_instruct-tts-1.wav
- instruct-tts_instruct-tts-2.wav
- instruct-tts_instruct-tts-3.wav
- instruct-tts_instruct-tts-4.wav
- nonverbal-edit_en-c.wav
- nonverbal-edit_en-d.wav
- nonverbal-edit_zh-a.wav
- nonverbal-edit_zh-b.wav
- pitch-edit_pitch-1.wav
- pitch-edit_pitch-2.wav
- separate-speech_en-1.wav
- separate-speech_en-2.wav
- separate-speech_zh-1.wav
- separate-speech_zh-2.wav
- speed-edit_speed-1.wav
- speed-edit_speed-2.wav
- super-resolution_sr-en-1.wav
- super-resolution_sr-en-2.wav
- super-resolution_sr-zh-1.wav
- super-resolution_sr-zh-2.wav
- vocal-edit_vocaledit-en-1.wav
- vocal-edit_vocaledit-en-2.wav
- vocal-edit_vocaledit-zh-1.wav
- vocal-edit_vocaledit-zh-2.wav
- voice-edit_vc-1.wav
- voice-edit_vc-2.wav
- voice-edit_vc-3.wav
- voice-edit_vc-4.wav
- volume-edit_volume-1.wav
- volume-edit_volume-2.wav
- whisper-edit_wh-n2w-en.wav
- whisper-edit_wh-n2w-zh.wav
- whisper-edit_wh-w2n-en.wav
- whisper-edit_wh-w2n-zh.wav
- zero-shot-tts_zs-tts-1.wav
- zero-shot-tts_zs-tts-2.wav
- zero-shot-tts_zs-tts-3.wav
- zero-shot-tts_zs-tts-4.wav
- zh-1_content.wav
- zh-1_number.wav
- zh-2_content.wav
- zh-2_number.wav
- accent-edit_accent-dongbei.wav
- accent-edit_accent-fujian.wav
- accent-edit_accent-hubei.wav
- accent-edit_accent-hunan.wav
- accent-edit_accent-india.wav
- accent-edit_accent-japan.wav
- accent-edit_accent-sichuan.wav
- accent-edit_accent-tibetan.wav
- ce-en-1_add.wav
- ce-en-1_delete.wav
- ce-en-1_replace.wav
- ce-en-2_add.wav
- ce-en-2_delete.wav
- ce-en-2_replace.wav
- ce-zh-1_add.wav
- ce-zh-1_delete.wav
- ce-zh-1_replace.wav
- ce-zh-2_add.wav
- ce-zh-2_delete.wav
- ce-zh-2_replace.wav
- content-edit-speech_ce-en-1.wav
- content-edit-speech_ce-en-2.wav
- content-edit-speech_ce-zh-1.wav
- content-edit-speech_ce-zh-2.wav
- emotion-edit_emo-en-1.wav
- emotion-edit_emo-en-2.wav
- emotion-edit_emo-zh-1.wav
- emotion-edit_emo-zh-2.wav
- en-1_content.wav
- en-1_number.wav
- en-2_content.wav
- en-2_number.wav
- enhance-speech_se-en-1.wav
- enhance-speech_se-en-2.wav
- enhance-speech_se-zh-1.wav
- enhance-speech_se-zh-2.wav
- extract-vocals_ev-1.wav
- extract-vocals_ev-2.wav
- extract-vocals_ev-3.wav
- extract-vocals_ev-4.wav
- instruct-tts_instruct-tts-1.wav
- instruct-tts_instruct-tts-2.wav
- instruct-tts_instruct-tts-3.wav
- instruct-tts_instruct-tts-4.wav
- nonverbal-edit_en-c.wav
- nonverbal-edit_en-d.wav
- nonverbal-edit_zh-a.wav
- nonverbal-edit_zh-b.wav
- pitch-edit_pitch-1.wav
- pitch-edit_pitch-2.wav
- separate-speech_en-1.wav
- separate-speech_en-2.wav
- separate-speech_zh-1.wav
- separate-speech_zh-2.wav
- speed-edit_speed-1.wav
- speed-edit_speed-2.wav
- super-resolution_sr-en-1.wav
- super-resolution_sr-en-2.wav
- super-resolution_sr-zh-1.wav
- super-resolution_sr-zh-2.wav
- vocal-edit_vocaledit-en-1.wav
- vocal-edit_vocaledit-en-2.wav
- vocal-edit_vocaledit-zh-1.wav
- vocal-edit_vocaledit-zh-2.wav
- voice-edit_vc-1.wav
- voice-edit_vc-2.wav
- voice-edit_vc-3.wav
- voice-edit_vc-4.wav
- volume-edit_volume-1.wav
- volume-edit_volume-2.wav
- whisper-edit_wh-n2w-en.wav
- whisper-edit_wh-n2w-zh.wav
- whisper-edit_wh-w2n-en.wav
- whisper-edit_wh-w2n-zh.wav
- zero-shot-tts_zs-tts-1.wav
- zero-shot-tts_zs-tts-2.wav
- zero-shot-tts_zs-tts-3.wav
- zero-shot-tts_zs-tts-4.wav
- zh-1_content.wav
- zh-1_number.wav
- zh-2_content.wav
- zh-2_number.wav
- ALL_62_DEMOS_RTF_REPORT.md
- OPTIMIZATION_AND_QUANTIZATION_REPORT.md
- RTF_BENCHMARK_REPORT.md
- convert_to_mlx.py
- download_demo_assets.py
- download_models.py
- quantize_mlx.py
- rebuild_all_natural_speed.py
- run_62_base_benchmarks.py
- run_all_62_benchmarks.py
- run_auk_web_benchmarks.py
- serve_demo.py
- cfm.py
- flux2.py
- modules.py
- encoder.py
- activations.py
- modules.py
- vae.py
- __init__.py
- anchoring.py
- config.py
- infer.py
- test_anchoring.py
- test_base_model.py
- test_cfm.py
- test_dit.py
- test_isolation.py
- test_quantization.py
- test_vae.py
- data.json
- index.html
- .ai_memory.md
- .ai_project.md
- .gitignore
- index.html
- pyproject.toml
- README.md
# Use via CDN
jsDelivrjsDelivr serves any public GitHub repository as a CDN with zero setup. Pick a version and a file to get a ready-to-paste link and snippet.
Link
Example
// repository documentation
Was this content helpful?
(0 ratings)
