stopes
A library for preparing data for machine translation research (monolingual preprocessing, bitext mining, etc.) built by the FAIR NLLB team.
File Explorer
Download Latest Version (.zip)- deploy-site.yaml
- lint_and_tests.yaml
- test-deploy-site.yaml
- pull_request_template.md
- annotate_hallucination_mitigation_v7_stacked.tsv
- guerreiro2022_corpus_w_annotations.csv
- 01_Detection.ipynb
- 02_Detection_analysis.ipynb
- 03_Mitigation.ipynb
- 04_Mitigation_more_hypotheses.ipynb
- 05_Mitigation_analysis.ipynb
- README.md
- requirements.txt
- nllb_demo.yaml
- compute_nllb_alti.py
- conf.yaml
- download_nllb.sh
- test_input.tsv
- test_input_de.tsv
- test_output.tsv
- test_output_alignments.jsonl
- .gitignore
- README.md
- __init__.py
- att_maps_compute.py
- att_maps_recombine.py
- optimal_transport_scoring.py
- compute_detection_scores.py
- evaluation_utils.py
- example_translation_script.sh
- LICENSE
- README.md
- reproduce_evaluation.py
- requirements-detection.txt
- requirements.txt
- eval_blaser.yaml
- launcher
- .gitignore
- mk_manifest.py
- prepare.sh
- README.md
- .gitignore
- prepare.sh
- README.md
- README.md
- flores_200_langs.json
- 00_compile_toxicity_stats.py
- 00c_plot_toxicity_per_lang.py
- 01_sample_high_risk_translations.py
- 02_count_toxicity_sources.py
- 02b_plot_alignment_type_breakdown.py
- 03_measure_source_contributions.py
- 03b_make_toxicity_heatmap.py
- README.md
- util.py
- README.md
- ETOX example calls.ipynb
- etox.py
- README.md
- README.md
- requirements.txt
- __init__.py
- registry.py
- stopes_job.py
- submitit_slurm_job.py
- department.yaml
- projects.yaml
- employee.yaml
- __init__.py
- hello_world.py
- test_headers.py
- test_launcher.py
- test_modules.py
- test_registry.py
- test_utils.py
- __init__.py
- cache.py
- launcher.py
- stopes_module.py
- utils.py
- __init__.py
- align.py
- alti_metrics_utils.py
- file_utils.py
- nllb_alti_detector.py
- __init__.py
- multilingual_transformer_wrapper.py
- transformer_wrapper.py
- utils.py
- __init__.py
- LICENSE.md
- README.md
- audio_comparator.py
- comparator_training.py
- contrastive_training.py
- prosody_probing.py
- README.md
- score.yaml
- train.yaml
- __init__.py
- blaser2.py
- model.py
- README.md
- score.py
- train.py
- utils.py
- test_emphasis_alignment.py
- test_force_align.py
- test_pause_alignment.py
- test_utterance.py
- __init__.py
- annotate_utterances.py
- compare_utterances.py
- ctc_forced_aligner.py
- emphasis_detection.py
- forced_aligner.py
- forced_aligner_utils.py
- phonemization.py
- README.md
- unity2_forced_aligner_f1.py
- unity2_forced_aligner_f2.py
- utterance.py
- toxicity_list.py
- ecapa.py
- README.md
- valle_sv.py
- vocal_style_sim_module.py
- vocal_style_sim_tool.py
- awesome_align_wrapper.py
- alignment_utils.py
- __init__.py
- README.md
- __init__.py
- merge_faiss_indexes.py
- populate_faiss_index.py
- sample_embedding_module.py
- train_faiss_index_module.py
- train_index.py
- __init__.py
- calculate_distances.py
- calculate_distances_utils.py
- count_lines.py
- merge_shards.py
- mine_bitext_indexes.py
- mine_bitext_indexes_utils.py
- mine_bitext_sentences.py
- mine_bitext_sentences_utils.py
- __init__.py
- laser_scorer.py
- __init__.py
- blaser_module.py
- compare_audio_module.py
- generate_multi_bleu_detok_module.py
- sentence_transformers_similarity.py
- __init__.py
- monolingual_sort_dedup.py
- __init__.py
- bitext_processor.py
- encode_to_npy.py
- fairseq_binarizer_encoder.py
- hf_sentence_encoder.py
- laser_sentence_encoder.py
- line_processor.py
- mining_speech_encoder.py
- moses_cli_module.py
- multiproc_bitext_processor.py
- multiproc_fairseq_binarizer_encoder.py
- multiproc_line_processor.py
- preprocess_encode_module.py
- sonar_sentence_encoder.py
- sonar_text_embedding.py
- split_in_shards.py
- train_spm.py
- uromanize_cli_module.py
- wav2vec_laser_speech_encoder.py
- __init__.py
- data.py
- models.py
- shas.py
- README.md
- video_segment_aligner.py
- video_segmentor.py
- video_utils.py
- __init__.py
- asr.py
- utils.py
- __init__.py
- audio_load_utils.py
- audio_zip.py
- denoise.py
- postprocess.py
- segment_and_lid.py
- segment_dataset.py
- shas_segment_audio.py
- speech_kmeans.py
- speech_units.py
- speechbrain_lid.py
- utils.py
- vad.py
- vad_segment_audio.py
- vad_trim_audio.py
- whisper.py
- __init__.py
- test_audiozip.py
- test_embedding_utils.py
- test_laser_scorer.py
- test_mine_index_utils.py
- test_modules_utils.py
- test_moses_cli.py
- test_partitioned_data_mapper.py
- test_populate_index_port.py
- test_sample_embedding.py
- test_speech_utils.py
- test_split_merge_langs.py
- test_split_mined_tsv.py
- test_text_input.py
- test_train_index_port.py
- __init__.py
- fairseq_generate.py
- moses-config.lowercase
- __init__.py
- __main__.py
- nmt_bitext_eval_utils.py
- partitioned_data_mapper.py
- train_fairseq_module.py
- base.yaml
- fairseq_binarize.yaml
- standard_conf.yaml
- base.yaml
- AutoPCP_multilingual_v2.yaml
- base.yaml
- count_lines.yaml
- denoise.yaml
- mining_speech_encoder.yaml
- sonar2_speech_encoder.yaml
- laser.yaml
- sonar.yaml
- hf_encoder.yaml
- laser2_encoder.yaml
- laser3_encoder.yaml
- sonar_encoder.yaml
- encode.yaml
- hf_labse.yaml
- hf_roberta_large.yaml
- huggingface.yaml
- laser2.yaml
- laser3.yaml
- preproc_and_encode.yaml
- sonar.yaml
- generate_multi_bleu_detok.yaml
- ctc_wav2vec2-xlsr-multilingual-56.yaml
- fairseq2_nar_t2u_aligner.yaml
- standard_conf.yaml
- file_cache.yaml
- local.yaml
- submitit.yaml
- merge_faiss.yaml
- base.yaml
- base.yaml
- standard_conf.yaml
- standard_conf.yaml
- populate_faiss.yaml
- demo.yaml
- demo_speechmine.yaml
- base.yaml
- segment_and_lid.yaml
- segment_dataset.yaml
- shas.yaml
- speechbrain_lid.yaml
- vad.yaml
- vad_trim.yaml
- video_segment_alignement.yaml
- video_segmentation.yaml
- encodec_24khz.yaml
- encodec_48khz.yaml
- whisper.yaml
- base.yaml
- standard_conf.yaml
- transformer.yaml
- nmt.yaml
- train_faiss.yaml
- standard_conf.yaml
- standard_conf.yaml
- base.yaml
- __init__.py
- bitext_eval.yaml
- dedup_local_and_global.yaml
- dedup_single_file.yaml
- demojize.yaml
- extract_meta.yaml
- global_mining.yaml
- launch_conf.yaml
- mine_added_toxicity.yaml
- nmt_bitext_eval.yaml
- shard_and_shuffle.yaml
- .gitignore
- __init__.py
- added_toxicity_mining.py
- bitext_eval.py
- dedup_local_and_global.py
- dedup_single_file.py
- DemojizeLineProc.py
- ExtractMetaLineProc.py
- global_mining_pipeline.py
- nmt_bitext_eval.py
- README.md
- requirements.txt
- shard_and_shuffle.py
- default.yaml
- default.yaml
- default.yaml
- default.yaml
- default.yaml
- default.yaml
- default.yaml
- transformer.yaml
- default.yaml
- distillation.yaml
- launcher
- __init__.py
- distillation_bitext_processor.py
- distillation_pipeline.py
- speech_encoder.yaml
- text_encoder.yaml
- eval_blaser.yaml
- launcher
- __init__.py
- eval_blaser.py
- example.yaml
- __init__.py
- base.py
- dedup.py
- laser.py
- length.py
- lid.py
- toxicity.py
- compute_length_factors.py
- populate_data_conf.py
- __init__.py
- configs.py
- dataset.py
- filter.py
- README.md
- utils.py
- file_cache.yaml
- local.yaml
- submitit.yaml
- m4t_eval.yaml
- config_def.py
- m4t_eval.py
- README.md
- launcher
- monolingual.yaml
- bod_sentenizer.py
- LICENSE
- urdu_sentenizer.py
- __init__.py
- predict_lid.py
- predict_script.py
- remove_regex.py
- sentence_split.py
- sort.py
- text_filter.py
- text_normalizer.py
- word_tokenization.py
- .gitignore
- __init__.py
- dedup_files.py
- language_equivalences.tsv
- language_scripts_200.tsv
- monolingual_line_processor.py
- monolingual_pipeline.py
- README.md
- requirements.txt
- test_mono.py
- both.yaml
- neither.yaml
- default.yaml
- default.yaml
- default.yaml
- default.yaml
- default.yaml
- launcher
- prepare_data.yaml
- __init__.py
- binarize.py
- build_vocab.py
- configs.py
- dedup_sharding.py
- prepare_data.py
- README.md
- retrieve_data.py
- sample_corpus.py
- validate.py
- audio_zip.yaml
- launcher
- uromanization.yaml
- wav2vec_asr.yaml
- whisper_asr.yaml
- wav2vec_asr.py
- whisper_asr.py
- launcher
- speech_kmeans.yaml
- speech_kmeans.py
- launcher
- speech_laser_embeddings.yaml
- README.md
- speech_laser_embeddings.py
- utils.py
- test_dummy_speech_encoder.yaml
- test_numbers_encoder.yaml
- __init__.py
- flat.yaml
- nested.yaml
- __init__.py
- test_configs.py
- test_global_mining.py
- example.yaml
- launcher
- __init__.py
- README.md
- translation_pipeline.py
- __init__.py
- __init__.py
- base_decoder.py
- decoder_config.py
- flashlight_decoder.py
- viterbi_decoder.py
- __init__.py
- notebook.py
- tokenizer_demo.ipynb
- __init__.py
- config.yaml
- partial_config.yaml
- test_speech_tokenizer.py
- __init__.py
- asr.py
- encodec.py
- extract_wav.py
- kmeans.py
- README.md
- tokenizers.py
- __init__.py
- api.py
- constants.py
- fileviewer.py
- query_types.py
- static.py
- dev.env
- install_backend.sh
- main.py
- prod.env
- requirements.txt
- requirements_lint.txt
- run_backend.sh
- uvicorn_logging_config.yml
- favicon.ico
- index.html
- logo.png
- robots.txt
- area_constructor.ts
- audioquery_constructor.ts
- PlayArea.css
- PlayArea.tsx
- WaveSurfer.css
- WaveSurfer.jsx
- error-page.jsx
- index.js
- spinner.css
- spinner.tsx
- table.css
- config.js
- audio.ts
- audiosearch.ts
- mining_result.ts
- api.ts
- index.d.ts
- LinesSelector.tsx
- Pagination.tsx
- Row.tsx
- FileExplorerHelp.tsx
- Table.tsx
- FileExplorer.tsx
- App.css
- App.test.js
- App.tsx
- index.css
- index.js
- reportWebVitals.js
- setupTests.js
- style.css
- .gitignore
- npm_update_user_only.sh
- package-lock.json
- package.json
- README.md
- tsconfig.json
- .gitignore
- package-lock.json
- package.json
- Readme.md
- __init__.py
- abstract_shards.py
- hf_shards.py
- json_shards.py
- parquet_shards.py
- text_shards.py
- conftest.py
- test_data_utils.py
- test_file_chunker_utils.py
- test_parquet_dataloader.py
- test_parquet_shards.py
- test_shards.py
- test_text_shards.py
- __init__.py
- utils.py
- words.py
- __init__.py
- __init__.py
- cleaners.py
- cmn.py
- sentence_split.py
- __init__.py
- aligner_utils.py
- arrow_utils.py
- asr_eval_utils.py
- cache.py
- checkpoint_utils.py
- config_utils.py
- data_utils.py
- demojizer.py
- embedding_utils.py
- file_chunker_utils.py
- language_codes.py
- map_token_lang.tsv
- math_utils.py
- mining_utils.py
- parquet_dataloader.py
- test_hf_shards.py
- test_json_shards.py
- web.py
- __init__.py
- hub.py
- _category_.json
- alti.md
- blaser.md
- _category_.json
- distillation.md
- expressive_alignments.md
- global_mining.md
- monolingual.md
- speech_mining.md
- _category_.json
- checkpointing.md
- debugging.md
- dynamic.md
- _category_.json
- cache.md
- configuration.md
- index.md
- module.md
- pipelining.md
- quickstart.md
- custom.css
- index.js
- styles.module.css
- driving.png
- large.png
- meta.png
- nllb.png
- stopes.png
- banner.png
- favicon.ico
- logo.svg
- meta_opensource_logo.svg
- meta_opensource_logo_negative.svg
- modules.svg
- pipelines.svg
- shovel.svg
- .nojekyll
- .eslintrc.js
- .gitignore
- .prettierignore
- .prettierrc
- .stylelintrc.js
- babel.config.js
- docusaurus.config.js
- package-lock.json
- package.json
- README.md
- sidebars.js
- .gitignore
- .pre-commit-config.yaml
- CHANGELOG.md
- CODE_OF_CONDUCT.md
- CONTRIBUTING.md
- LICENSE.md
- package-lock.json
- package.json
- pyproject.toml
- README.md
# Installation Guide
git clone https://github.com/facebookresearch/stopes
Downloads the entire project code from GitHub to your computer.
cd stopes
Moves into the project folder you just downloaded.
2. Official Install Script
Easy Recommended- Python 3 Python is required to use pip.
pip install fairseq==0.12.2
Installs the package published on PyPI directly โ no need to clone the source.
pip install -v --no-cache-dir --global-option="--cpp_ext" --global-option="--cuda_ext" \
Installs the package published on PyPI directly โ no need to clone the source.
Pulled directly from this repo's README.
3. Node.js
Easynpm install
Downloads and installs the libraries listed in package.json.
npm start
Starts the development/run server.
4. Python
Easypip install fairseq==0.12.2
Installs the package published on PyPI directly โ no need to clone the source.
pip install -e '.[dev,mono,mining]'
Installs the Python libraries listed in requirements.txt (or similar).
pip install -v --no-cache-dir --global-option="--cpp_ext" --global-option="--cuda_ext" \
Installs the package published on PyPI directly โ no need to clone the source.
Pulled directly from this repo's README.
