OSUM
OSUM & OSUM-EChat, open speech understanding model and empathetic spoken chatbot based on it, open-sourced by ASLP@NPU.
파일 탐색기
- demo_cn.png
- demo_en.png
- dual_think_cn.png
- dual_think_en.png
- SUM.png
- system.png
- table1.png
- table2.png
- table3.png
- table4.png
- table5.png
- Architecture.png
- ASLP.jpg
- introduction.md
- OSUM_communicating.jpg
- radar.jpg
- radar.pdf
- radar.png
- res_asr.jpg
- res_asr.png
- res_multi.jpg
- res_multi.png
- SUM.png
- system.png
- wechat.png
- wenet.png
- do_convert_dir_to_pt.py
- handle_data_for_weight.py
- 实验室.png
- config_llm_huawei_base-version.yaml
- data_config_huawei.yaml
- ds_stage1.json
- ds_stage2.json
- prompt_config.yaml
- do_docode.sh
- remove_dup_utts.sh
- compile_lexicon_token_fst.sh
- ctc_token_fst.py
- ctc_token_fst_compact.py
- ctc_token_fst_corrected.py
- make_tlg.sh
- prepare_dict.py
- rnnt_token_fst.py
- make_hlg.sh
- prepare_char.py
- prepare_mmi.sh
- performance-ws.py
- alignment.sh
- analyze_dataset.py
- cmvn_kaldi2json.py
- combine_data.sh
- compute-acc.py
- compute-cer.py
- compute-wer.py
- compute_cmvn_stats.py
- compute_fbank_feats.py
- compute_shard_cmvn_stats.py
- copy_data_dir.sh
- decode.sh
- extract_shard_data.py
- feat_to_shape.sh
- fix_data_dir.sh
- flake8_hook.py
- format_data.sh
- install_srilm.sh
- latency_metrics.py
- make_raw_list.py
- make_shard_list.py
- make_shard_list_osum.py
- merge_scp2txt.py
- onnx2horizonbin.py
- parse_options.sh
- perturb_data_dir_speed.sh
- reduce_data_dir.sh
- remove_longshortdata.py
- segment.py
- setup_anaconda.sh
- sph2wav.sh
- ssh_launcher.py
- subset_data_dir.sh
- text2token.py
- validate_data_dir.sh
- wav2dur.py
- wav_to_duration.sh
- alignment.py
- average_model.py
- export_ipex.py
- export_jit.py
- export_onnx_bpu.py
- export_onnx_cpu.py
- export_onnx_gpu.py
- recognize.py
- recognize4llmasr.py
- recognize_onnx_gpu.py
- train.py
- __init__.py
- cgmlp.py
- encoder.py
- encoder_layer.py
- __init__.py
- hub.py
- model.py
- paraformer_model.py
- transcribe.py
- asr_model_ctl.py
- encoder.py
- processor.py
- __init__.py
- datapipes.py
- dataset.py
- kaldi_io.py
- wav_distortion.py
- encoder.py
- encoder_layer.py
- __init__.py
- attention.py
- convolution.py
- encoder.py
- encoder_layer.py
- subsampling.py
- __init__.py
- config.yaml
- layers.py
- utils.py
- __init__.py
- model.py
- causallm_model.py
- decoder.py
- sampler.py
- __init__.py
- downsampler.py
- init_llmasr.py
- llmasr_model.py
- utils4llmasr.py
- __init__.py
- attention.py
- cif.py
- convert_paraformer_to_wenet_config_and_ckpt.py
- embedding.py
- layers.py
- paraformer.py
- search.py
- subsampling.py
- __init__.py
- attention.py
- conv2d.py
- convolution.py
- encoder.py
- encoder_layer.py
- positionwise_feed_forward.py
- subsampling.py
- bestrq_model.py
- mask.py
- convert_w2vbert_to_wenet_config_and_ckpt.py
- w2vbert_model.py
- quantizer.py
- wav2vec2_model.py
- init_dataset.py
- init_model.py
- __init__.py
- base_tokenizer.py
- bpe_tokenizer.py
- char_tokenizer.py
- hugging_face_tokenizer.py
- paraformer_tokenizer.py
- tokenize_utils.py
- whisper_tokenizer.py
- greedy_search.py
- prefix_beam_search.py
- __init__.py
- joint.py
- predictor.py
- transducer.py
- __init__.py
- asr_model.py
- attention.py
- cmvn.py
- convolution.py
- ctc.py
- decoder.py
- decoder_layer.py
- embedding.py
- encoder.py
- encoder_layer.py
- label_smoothing_loss.py
- norm.py
- positionwise_feed_forward.py
- search.py
- subsampling.py
- swish.py
- __init__.py
- checkpoint.py
- class_utils.py
- cmvn.py
- common.py
- config.py
- context_graph.py
- ctc_utils.py
- executor.py
- file_utils.py
- fsdp_utils.py
- init_dataset.py
- init_model.py
- init_tokenizer.py
- mask.py
- rope_utils.py
- scheduler.py
- train_utils.py
- __init__.py
- convert_whisper_to_wenet_config_and_ckpt.py
- whisper.py
- whisper_with_clap.py
- __init__.py
- infer.sh
- infer_gradio.py
- infer_runtime.py
- path.sh
- README.md
- README_CN.md
- requirements.txt
- run_huawei_2p_master.sh
- run_huawei_2p_rank1.sh
- run_huawei_2p_rank2.sh
- __init__.py
- dataset_no_wav.py
- processor_no_wav.py
- combines_list.txt
- combines_tar_root.txt
- data.list
- do_convert_tar_type2tardata_combine_type.py
- shards_000000000.list
- random.wav
- data.list
- do_make_shard_from_raw.py
- shards_000000000.tar
- shards_000000000.tar.finished
- shards_list.txt
- __init__.py
- convert_ckpt_dir_to_pt.py
- load_combine_type_yaml.py
- utils4infer.py
- ct_config.yaml
- data_tmp.yaml
- ds_stage2.json
- empty.yaml
- prompt_config.yaml
- cumstom_stop_criteria.py
- custom_speech_ngram_blocking.py
- custom_speech_repetition_penalty.py
- modelling_fm_infer_gpu.py
- modelling_qwen2_infer_gpu.py
- utils.py
- hq_1.wav
- prompt.wav
- prompt2.wav
- 实验室.png
- average_model.py
- export_jit.py
- export_onnx.py
- export_trt.sh
- inference.py
- train.py
- __init__.py
- cosyvoice.py
- frontend.py
- model.py
- __init__.py
- dataset.py
- processor.py
- decoder.py
- flow.py
- flow_matching.py
- length_regulator.py
- discriminator.py
- f0_predictor.py
- generator.py
- hifigan.py
- llm.py
- multilingual_zh_ja_yue_char_del.tiktoken
- tokenizer.py
- __init__.py
- activation.py
- attention.py
- convolution.py
- decoder.py
- decoder_layer.py
- embedding.py
- encoder.py
- encoder_layer.py
- label_smoothing_loss.py
- positionwise_feed_forward.py
- subsampling.py
- upsample_encoder.py
- __init__.py
- class_utils.py
- common.py
- executor.py
- file_utils.py
- frontend_utils.py
- losses.py
- mask.py
- scheduler.py
- test.ipynb
- train_utils.py
- __init__.py
- test2cosyvoice1-25hz_0_gxl.wav
- test2cosyvoice1-25hz_1_gxl.wav
- test2cosyvoice1-25hz_2_gxl.wav
- test2cosyvoice1-25hz_3_gxl.wav
- test2cosyvoice1-25hz_4_gxl.wav
- test2cosyvoice1-25hz_5_gxl.wav
- codecov.yml
- dependabot.yml
- PULL_REQUEST_TEMPLATE.md
- release-drafter.yml
- default.yaml
- model_checkpoint.yaml
- model_summary.yaml
- none.yaml
- rich_progress_bar.yaml
- hi-fi_en-US_female.yaml
- ljspeech.yaml
- vctk.yaml
- default.yaml
- fdr.yaml
- limit.yaml
- overfit.yaml
- profiler.yaml
- hifi_dataset_piper_phonemizer.yaml
- ljspeech.yaml
- ljspeech_min_memory.yaml
- multispeaker.yaml
- default.yaml
- mnist_optuna.yaml
- default.yaml
- .gitkeep
- aim.yaml
- comet.yaml
- csv.yaml
- many_loggers.yaml
- mlflow.yaml
- neptune.yaml
- tensorboard.yaml
- wandb.yaml
- default.yaml
- default.yaml
- default.yaml
- adam.yaml
- matcha.yaml
- default.yaml
- cpu.yaml
- ddp.yaml
- ddp_sim.yaml
- default.yaml
- gpu.yaml
- mps.yaml
- __init__.py
- eval.yaml
- train.yaml
- __init__.py
- __init__.py
- text_mel_datamodule.py
- __init__.py
- config.py
- denoiser.py
- env.py
- LICENSE
- meldataset.py
- models.py
- README.md
- xutils.py
- __init__.py
- decoder.py
- flow_matching.py
- text_encoder.py
- transformer.py
- __init__.py
- baselightningmodule.py
- matcha_tts.py
- __init__.py
- export.py
- infer.py
- __init__.py
- cleaners.py
- numbers.py
- symbols.py
- __init__.py
- core.pyx
- setup.py
- __init__.py
- audio.py
- generate_data_statistics.py
- instantiators.py
- logging_utils.py
- model.py
- pylogger.py
- rich_utils.py
- utils.py
- __init__.py
- app.py
- cli.py
- train.py
- VERSION
- .gitkeep
- schedule.sh
- .env.example
- .gitignore
- .pre-commit-config.yaml
- .project-root
- .pylintrc
- LICENSE
- Makefile
- MANIFEST.in
- pyproject.toml
- README.md
- requirements.txt
- setup.py
- synthesis.ipynb
- extract_embedding.py
- extract_speech_token.py
- make_parquet_list.py
- __init__.py
- fusion_result.json
- infer_cosyvoice1_gxl.py
- requirements.txt
- average_model.py
- train.py
- processor.py
- processor_language_think.py
- processor_tag_think.py
- __init__.py
- dataset.py
- __init__.py
- downsampler.py
- init_llmasr.py
- llmasr_model_instruct_version.py
- tmp.py
- utils4llmasr.py
- wav_instrcut_tools.py
- __init__.py
- base_tokenizer.py
- bpe_tokenizer.py
- char_tokenizer.py
- hugging_face_tokenizer.py
- paraformer_tokenizer.py
- tokenize_utils.py
- whisper_tokenizer.py
- __init__.py
- asr_model.py
- attention.py
- cmvn.py
- convolution.py
- ctc.py
- decoder.py
- decoder_layer.py
- embedding.py
- encoder.py
- encoder_layer.py
- label_smoothing_loss.py
- norm.py
- positionwise_feed_forward.py
- search.py
- subsampling.py
- swish.py
- __init__.py
- checkpoint.py
- class_utils.py
- cmvn.py
- common.py
- config.py
- context_graph.py
- ctc_utils.py
- executor.py
- file_utils.py
- fsdp_utils.py
- init_dataset.py
- init_model.py
- init_tokenizer.py
- mask.py
- rope_utils.py
- scheduler.py
- train_utils.py
- __init__.py
- convert_whisper_to_wenet_config_and_ckpt.py
- whisper.py
- whisper_with_clap.py
- __init__.py
- do_test_dataloader.py
- infer_gradio.py
- infer_multiturn.py
- infer_runtime.py
- infer_with_shards_or_raw.py
- infer_with_shards_or_raw.sh
- README.md
- README_CN.md
- requirements.txt
- tmp.py
- train.sh
- .gitignore
- LICENSE.txt
- README.md
- README_CN.md
- README_JP.md
# CDN으로 사용하기
jsDelivrjsDelivr는 공개 GitHub 리포지토리를 별도 설정 없이 CDN으로 즉시 서빙합니다. 버전과 파일을 고르면 웹페이지에 바로 붙일 수 있는 링크와 예시 코드가 만들어집니다.
링크
예시
// repository documentation
Was this content helpful?
(0 ratings)
